fixup! Fix load explosion: use subprocess for binary data plots to avoid thread conflict

* Seems we don't have to set so many variables, `OMP_NUM_THREADS` is enough. Test: Annotate the code for setting other environment variables. It runs normally.
Fix load explosion: use subprocess for binary data plots to avoid thread conflict
2026-02-09 23:00:17 +08:00 · 2026-02-09 21:40:27 +08:00 · 2026-02-09 21:36:45 +08:00 · 2026-02-09 15:13:18 +08:00 · 2026-02-08 16:46:44 +08:00 · 2026-02-08 16:19:13 +08:00
15 changed files with 2012 additions and 833 deletions
--- a/AMSS_NCKU_ABEtest.py
+++ b/AMSS_NCKU_ABEtest.py
@@ -1,447 +0,0 @@
-
-##################################################################
-##
-## AMSS-NCKU ABE Test Program (Skip TwoPuncture if data exists)
-## Modified from AMSS_NCKU_Program.py
-## Author: Xiaoqu
-## Modified: 2026/02/01
-##
-##################################################################
-
-
-##################################################################
-
-## Print program introduction
-
-import print_information
-
-print_information.print_program_introduction()
-
-##################################################################
-
-import AMSS_NCKU_Input as input_data
-
-##################################################################
-
-## Create directories to store program run data
-
-import os
-import shutil
-import sys
-import time
-
-## Set the output directory according to the input file
-File_directory = os.path.join(input_data.File_directory)
-
-## Check if output directory exists and if TwoPuncture data is available
-#skip_twopuncture = False
-skip_twopuncture = True
-output_directory = os.path.join(File_directory, "AMSS_NCKU_output")
-binary_results_directory = os.path.join(output_directory, input_data.Output_directory)
-
-if os.path.exists(File_directory):
-    print( " Output directory already exists." )
-    print()
-    '''
-    # Check if TwoPuncture initial data files exist
-    if (input_data.Initial_Data_Method == "Ansorg-TwoPuncture"):
-        twopuncture_output = os.path.join(output_directory, "TwoPunctureABE")
-        input_par = os.path.join(output_directory, "input.par")
-
-        if os.path.exists(twopuncture_output) and os.path.exists(input_par):
-            print( " Found existing TwoPuncture initial data." )
-            print( " Do you want to skip TwoPuncture phase and reuse existing data?" )
-            print( " Input 'skip' to skip TwoPuncture and start ABE directly" )
-            print( " Input 'regenerate' to regenerate everything from scratch" )
-            print()
-            
-            while True:
-                try:
-                    inputvalue = input()
-                    if ( inputvalue == "skip" ):
-                        print( " Skipping TwoPuncture phase, will reuse existing initial data." )
-                        print()
-                        skip_twopuncture = True
-                        break
-                    elif ( inputvalue == "regenerate" ):
-                        print( " Regenerating everything from scratch." )
-                        print()
-                        skip_twopuncture = False
-                        break
-                    else:
-                        print( " Please input 'skip' or 'regenerate'." )
-                except ValueError:
-                    print( " Please input 'skip' or 'regenerate'." )
-            
-        else:
-            print( " TwoPuncture initial data not found, will regenerate everything." )
-            print()
-'''
-    # If not skipping, remove and recreate directory
-    if not skip_twopuncture:
-        shutil.rmtree(File_directory, ignore_errors=True)
-        os.mkdir(File_directory)
-        os.mkdir(output_directory)
-        os.mkdir(binary_results_directory)
-        figure_directory = os.path.join(File_directory, "figure")
-        os.mkdir(figure_directory)
-        shutil.copy("AMSS_NCKU_Input.py", File_directory)
-        print( " Output directory has been regenerated." )
-        print()
-else:
-    # Create fresh directory structure
-    os.mkdir(File_directory)
-    shutil.copy("AMSS_NCKU_Input.py", File_directory)
-    os.mkdir(output_directory)
-    os.mkdir(binary_results_directory)
-    figure_directory = os.path.join(File_directory, "figure")
-    os.mkdir(figure_directory)
-    print( " Output directory has been generated." )
-    print()
-
-# Ensure figure directory exists
-figure_directory = os.path.join(File_directory, "figure")
-if not os.path.exists(figure_directory):
-    os.mkdir(figure_directory)
-
-##################################################################
-
-## Output related parameter information
-
-import setup
-
-## Print and save input parameter information
-setup.print_input_data( File_directory )
-
-if not skip_twopuncture:
-    setup.generate_AMSSNCKU_input()
-
-setup.print_puncture_information()
-
-
-##################################################################
-
-## Generate AMSS-NCKU program input files based on the configured parameters
-
-if not skip_twopuncture:
-    print()
-    print( " Generating the AMSS-NCKU input parfile for the ABE executable." )
-    print()
-
-    ## Generate cgh-related input files from the grid information
-
-    import numerical_grid
-
-    numerical_grid.append_AMSSNCKU_cgh_input()
-
-    print()
-    print( " The input parfile for AMSS-NCKU C++ executable file ABE has been generated." )
-    print( " However, the input relevant to TwoPuncture need to be appended later." )
-    print()
-
-
-##################################################################
-
-## Plot the initial grid configuration
-
-if not skip_twopuncture:
-    print()
-    print( " Schematically plot the numerical grid structure." )
-    print()
-
-    import numerical_grid
-    numerical_grid.plot_initial_grid()
-
-
-##################################################################
-
-## Generate AMSS-NCKU macro files according to the numerical scheme and parameters
-
-if not skip_twopuncture:
-    print()
-    print( " Automatically generating the macro file for AMSS-NCKU C++ executable file ABE " )
-    print( " (Based on the finite-difference numerical scheme) " )
-    print()
-
-    import generate_macrodef
-
-    generate_macrodef.generate_macrodef_h()
-    print( " AMSS-NCKU macro file macrodef.h has been generated. " )
-
-    generate_macrodef.generate_macrodef_fh()
-    print( " AMSS-NCKU macro file macrodef.fh has been generated. " )
-
-
-##################################################################
-
-# Compile the AMSS-NCKU program according to user requirements
-# NOTE: ABE compilation is always performed, even when skipping TwoPuncture
-
-print()
-print( " Preparing to compile and run the AMSS-NCKU code as requested " )
-print( " Compiling the AMSS-NCKU code based on the generated macro files " )
-print()
-
-AMSS_NCKU_source_path = "AMSS_NCKU_source"
-AMSS_NCKU_source_copy = os.path.join(File_directory, "AMSS_NCKU_source_copy")
-
-## If AMSS_NCKU source folder is missing, create it and prompt the user
-if not os.path.exists(AMSS_NCKU_source_path):
-    os.makedirs(AMSS_NCKU_source_path)
-    print( " The AMSS-NCKU source files are incomplete; copy all source files into ./AMSS_NCKU_source. " )
-    print( " Press Enter to continue. " )
-    inputvalue = input()
-
-# Copy AMSS-NCKU source files to prepare for compilation
-# If skipping TwoPuncture and source_copy already exists, remove it first
-if skip_twopuncture and os.path.exists(AMSS_NCKU_source_copy):
-    shutil.rmtree(AMSS_NCKU_source_copy)
-
-shutil.copytree(AMSS_NCKU_source_path, AMSS_NCKU_source_copy)
-
-# Copy the generated macro files into the AMSS_NCKU source folder
-if not skip_twopuncture:
-    macrodef_h_path  = os.path.join(File_directory, "macrodef.h")
-    macrodef_fh_path = os.path.join(File_directory, "macrodef.fh")
-else:
-    # When skipping TwoPuncture, use existing macro files from previous run
-    macrodef_h_path  = os.path.join(File_directory, "macrodef.h")
-    macrodef_fh_path = os.path.join(File_directory, "macrodef.fh")
-
-shutil.copy2(macrodef_h_path,  AMSS_NCKU_source_copy)
-shutil.copy2(macrodef_fh_path, AMSS_NCKU_source_copy)
-
-# Compile related programs
-import makefile_and_run
-
-## Change working directory to the target source copy
-os.chdir(AMSS_NCKU_source_copy)
-
-## Build the main AMSS-NCKU executable (ABE or ABEGPU)
-makefile_and_run.makefile_ABE()
-
-## If the initial-data method is Ansorg-TwoPuncture, build the TwoPunctureABE executable
-## Only build TwoPunctureABE if not skipping TwoPuncture phase
-if (input_data.Initial_Data_Method == "Ansorg-TwoPuncture" ) and not skip_twopuncture:
-    makefile_and_run.makefile_TwoPunctureABE()
-
-## Change current working directory back up two levels
-os.chdir('..')
-os.chdir('..')
-
-print()
-
-##################################################################
-
-## Copy the AMSS-NCKU executable (ABE/ABEGPU) to the run directory
-
-if (input_data.GPU_Calculation == "no"):
-    ABE_file = os.path.join(AMSS_NCKU_source_copy, "ABE")
-elif (input_data.GPU_Calculation == "yes"):
-    ABE_file = os.path.join(AMSS_NCKU_source_copy, "ABEGPU")
-
-if not os.path.exists( ABE_file ):
-    print()
-    print( " Lack of AMSS-NCKU executable file ABE/ABEGPU; recompile AMSS_NCKU_source manually. " )
-    print( " When recompilation is finished, press Enter to continue. " )
-    inputvalue = input()
-
-## Copy the executable ABE (or ABEGPU) into the run directory
-shutil.copy2(ABE_file, output_directory)
-
-## If the initial-data method is TwoPuncture, copy the TwoPunctureABE executable to the run directory
-## Only copy TwoPunctureABE if not skipping TwoPuncture phase
-if (input_data.Initial_Data_Method == "Ansorg-TwoPuncture" ) and not skip_twopuncture:
-    TwoPuncture_file = os.path.join(AMSS_NCKU_source_copy, "TwoPunctureABE")
-
-    if not os.path.exists( TwoPuncture_file ):
-        print()
-        print( " Lack of AMSS-NCKU executable file TwoPunctureABE; recompile TwoPunctureABE in AMSS_NCKU_source. " )
-        print( " When recompilation is finished, press Enter to continue. " )
-        inputvalue = input()
-
-    ## Copy the TwoPunctureABE executable into the run directory
-    shutil.copy2(TwoPuncture_file, output_directory)
-
-##################################################################
-
-## If the initial-data method is TwoPuncture, generate the TwoPuncture input files
-
-if (input_data.Initial_Data_Method == "Ansorg-TwoPuncture" ) and not skip_twopuncture:
-
-    print()
-    print( " Initial data is chosen as Ansorg-TwoPuncture" )
-    print()
-    
-    print()
-    print( " Automatically generating the input parfile for the TwoPunctureABE executable " )
-    print()
-    
-    import generate_TwoPuncture_input
-    
-    generate_TwoPuncture_input.generate_AMSSNCKU_TwoPuncture_input()
-    
-    print()
-    print( " The input parfile for the TwoPunctureABE executable has been generated. " )
-    print()
-    
-    ## Generated AMSS-NCKU TwoPuncture input filename
-    AMSS_NCKU_TwoPuncture_inputfile      = 'AMSS-NCKU-TwoPuncture.input'
-    AMSS_NCKU_TwoPuncture_inputfile_path = os.path.join( File_directory, AMSS_NCKU_TwoPuncture_inputfile )
- 
-    ## Copy and rename the file
-    shutil.copy2( AMSS_NCKU_TwoPuncture_inputfile_path, os.path.join(output_directory, 'TwoPunctureinput.par') )
-    
-    ## Run TwoPuncture to generate initial-data files
-    
-    start_time = time.time()  # Record start time
-
-    print()
-    print()
-    
-    ## Change to the output (run) directory
-    os.chdir(output_directory)
-
-    ## Run the TwoPuncture executable
-    import makefile_and_run
-    makefile_and_run.run_TwoPunctureABE()
-    
-    ## Change current working directory back up two levels
-    os.chdir('..')
-    os.chdir('..')
-
-elif (input_data.Initial_Data_Method == "Ansorg-TwoPuncture" ) and skip_twopuncture:
-    print()
-    print( " Skipping TwoPuncture execution, using existing initial data." )
-    print()
-    start_time = time.time()  # Record start time for ABE only
-else:
-    start_time = time.time()  # Record start time
-    
-##################################################################
-    
-## Update puncture data based on TwoPuncture run results
-
-if not skip_twopuncture:
-    import renew_puncture_parameter
-    renew_puncture_parameter.append_AMSSNCKU_BSSN_input(File_directory, output_directory)
-
-    ## Generated AMSS-NCKU input filename
-    AMSS_NCKU_inputfile      = 'AMSS-NCKU.input'
-    AMSS_NCKU_inputfile_path = os.path.join(File_directory, AMSS_NCKU_inputfile)
- 
-    ## Copy and rename the file
-    shutil.copy2( AMSS_NCKU_inputfile_path, os.path.join(output_directory, 'input.par') )
-
-    print()
-    print( " Successfully copy all AMSS-NCKU input parfile to target dictionary. " )
-    print()
-else:
-    print()
-    print( " Using existing input.par file from previous run." )
-    print()
-
-##################################################################
-
-## Launch the AMSS-NCKU program
-
-print()
-print()
-
-## Change to the run directory
-os.chdir( output_directory )
-
-import makefile_and_run
-makefile_and_run.run_ABE()
-
-## Change current working directory back up two levels
-os.chdir('..')
-os.chdir('..')
-
-end_time = time.time()
-elapsed_time = end_time - start_time
-
-##################################################################
-
-## Copy some basic input and log files out to facilitate debugging
-
-## Path to the file that stores calculation settings
-AMSS_NCKU_error_file_path = os.path.join(binary_results_directory, "setting.par")
-## Copy and rename the file for easier inspection
-shutil.copy( AMSS_NCKU_error_file_path, os.path.join(output_directory, "AMSSNCKU_setting_parameter") )
-
-## Path to the error log file
-AMSS_NCKU_error_file_path = os.path.join(binary_results_directory, "Error.log")
-## Copy and rename the error log
-shutil.copy( AMSS_NCKU_error_file_path, os.path.join(output_directory, "Error.log") )
-
-## Primary program outputs
-AMSS_NCKU_BH_data         = os.path.join(binary_results_directory, "bssn_BH.dat"        )
-AMSS_NCKU_ADM_data        = os.path.join(binary_results_directory, "bssn_ADMQs.dat"     )
-AMSS_NCKU_psi4_data       = os.path.join(binary_results_directory, "bssn_psi4.dat"      )
-AMSS_NCKU_constraint_data = os.path.join(binary_results_directory, "bssn_constraint.dat")
-## copy and rename the file
-shutil.copy( AMSS_NCKU_BH_data,         os.path.join(output_directory, "bssn_BH.dat"        ) )
-shutil.copy( AMSS_NCKU_ADM_data,        os.path.join(output_directory, "bssn_ADMQs.dat"     ) )
-shutil.copy( AMSS_NCKU_psi4_data,       os.path.join(output_directory, "bssn_psi4.dat"      ) )
-shutil.copy( AMSS_NCKU_constraint_data, os.path.join(output_directory, "bssn_constraint.dat") )
-
-## Additional program outputs
-if (input_data.Equation_Class == "BSSN-EM"):
-    AMSS_NCKU_phi1_data = os.path.join(binary_results_directory, "bssn_phi1.dat" )
-    AMSS_NCKU_phi2_data = os.path.join(binary_results_directory, "bssn_phi2.dat" )
-    shutil.copy( AMSS_NCKU_phi1_data, os.path.join(output_directory, "bssn_phi1.dat" ) )
-    shutil.copy( AMSS_NCKU_phi2_data, os.path.join(output_directory, "bssn_phi2.dat" ) )
-elif (input_data.Equation_Class == "BSSN-EScalar"):
-    AMSS_NCKU_maxs_data = os.path.join(binary_results_directory, "bssn_maxs.dat" )
-    shutil.copy( AMSS_NCKU_maxs_data, os.path.join(output_directory, "bssn_maxs.dat" ) )
-
-##################################################################
-
-## Plot the AMSS-NCKU program results
-
-print()
-print( " Plotting the txt and binary results data from the AMSS-NCKU simulation " )
-print()
-
-
-import plot_xiaoqu
-import plot_GW_strain_amplitude_xiaoqu
-
-## Plot black hole trajectory
-plot_xiaoqu.generate_puncture_orbit_plot(   binary_results_directory, figure_directory )
-plot_xiaoqu.generate_puncture_orbit_plot3D( binary_results_directory, figure_directory )
-
-## Plot black hole separation vs. time
-plot_xiaoqu.generate_puncture_distence_plot( binary_results_directory, figure_directory )
-
-## Plot gravitational waveforms (psi4 and strain amplitude)
-for i in range(input_data.Detector_Number):
-    plot_xiaoqu.generate_gravitational_wave_psi4_plot( binary_results_directory, figure_directory, i )
-    plot_GW_strain_amplitude_xiaoqu.generate_gravitational_wave_amplitude_plot( binary_results_directory, figure_directory, i )
-
-## Plot ADM mass evolution
-for i in range(input_data.Detector_Number):
-    plot_xiaoqu.generate_ADMmass_plot( binary_results_directory, figure_directory, i )
-
-## Plot Hamiltonian constraint violation over time
-for i in range(input_data.grid_level):
-    plot_xiaoqu.generate_constraint_check_plot( binary_results_directory, figure_directory, i )
-
-## Plot stored binary data
-plot_xiaoqu.generate_binary_data_plot( binary_results_directory, figure_directory )
-
-print()
-print( f" This Program Cost = {elapsed_time} Seconds " )
-print()
-
-
-##################################################################
-
-print()
-print( " The AMSS-NCKU-Python simulation is successfully finished, thanks for using !!! " )
-print()
-
-##################################################################
-
-
--- a/AMSS_NCKU_Program.py
+++ b/AMSS_NCKU_Program.py
@@ -8,6 +8,14 @@
 ##
 ##################################################################

+## Guard against re-execution by multiprocessing child processes.
+## Without this, using 'spawn' or 'forkserver' context would cause every
+## worker to re-run the entire script, spawning exponentially more
+## workers (fork bomb).
+if __name__ != '__main__':
+    import sys as _sys
+    _sys.exit(0)
+

 ##################################################################

@@ -424,26 +432,31 @@ print(

 import plot_xiaoqu
 import plot_GW_strain_amplitude_xiaoqu
+from parallel_plot_helper import run_plot_tasks_parallel
+
+plot_tasks = []

 ## Plot black hole trajectory
-plot_xiaoqu.generate_puncture_orbit_plot(   binary_results_directory, figure_directory )
-plot_xiaoqu.generate_puncture_orbit_plot3D( binary_results_directory, figure_directory )
+plot_tasks.append( ( plot_xiaoqu.generate_puncture_orbit_plot,   (binary_results_directory, figure_directory) ) )
+plot_tasks.append( ( plot_xiaoqu.generate_puncture_orbit_plot3D, (binary_results_directory, figure_directory) ) )

 ## Plot black hole separation vs. time
-plot_xiaoqu.generate_puncture_distence_plot( binary_results_directory, figure_directory )
+plot_tasks.append( ( plot_xiaoqu.generate_puncture_distence_plot, (binary_results_directory, figure_directory) ) )

 ## Plot gravitational waveforms (psi4 and strain amplitude)
 for i in range(input_data.Detector_Number):
-    plot_xiaoqu.generate_gravitational_wave_psi4_plot( binary_results_directory, figure_directory, i )
-    plot_GW_strain_amplitude_xiaoqu.generate_gravitational_wave_amplitude_plot( binary_results_directory, figure_directory, i )
+    plot_tasks.append( ( plot_xiaoqu.generate_gravitational_wave_psi4_plot, (binary_results_directory, figure_directory, i) ) )
+    plot_tasks.append( ( plot_GW_strain_amplitude_xiaoqu.generate_gravitational_wave_amplitude_plot, (binary_results_directory, figure_directory, i) ) )

 ## Plot ADM mass evolution
 for i in range(input_data.Detector_Number):
-    plot_xiaoqu.generate_ADMmass_plot( binary_results_directory, figure_directory, i )
+    plot_tasks.append( ( plot_xiaoqu.generate_ADMmass_plot, (binary_results_directory, figure_directory, i) ) )

 ## Plot Hamiltonian constraint violation over time
 for i in range(input_data.grid_level):
-    plot_xiaoqu.generate_constraint_check_plot( binary_results_directory, figure_directory, i )
+    plot_tasks.append( ( plot_xiaoqu.generate_constraint_check_plot, (binary_results_directory, figure_directory, i) ) )
+
+run_plot_tasks_parallel(plot_tasks)

 ## Plot stored binary data
 plot_xiaoqu.generate_binary_data_plot( binary_results_directory, figure_directory )
--- a/AMSS_NCKU_source/Parallel.C
+++ b/AMSS_NCKU_source/Parallel.C
@@ -3756,6 +3756,358 @@ void Parallel::Sync(MyList<Patch> *PatL, MyList<var> *VarList, int Symmetry)
  delete[] transfer_src;
  delete[] transfer_dst;
 }
+//
+// Async Sync: split into SyncBegin (initiate MPI) and SyncEnd (wait + unpack)
+// This allows overlapping MPI communication with computation.
+//
+static void transfer_begin(Parallel::TransferState *ts)
+{
+  int myrank;
+  MPI_Comm_rank(MPI_COMM_WORLD, &myrank);
+  int cpusize = ts->cpusize;
+
+  ts->reqs = new MPI_Request[2 * cpusize];
+  ts->stats = new MPI_Status[2 * cpusize];
+  ts->req_no = 0;
+  ts->send_data = new double *[cpusize];
+  ts->rec_data = new double *[cpusize];
+  int length;
+
+  for (int node = 0; node < cpusize; node++)
+  {
+    ts->send_data[node] = ts->rec_data[node] = 0;
+    if (node == myrank)
+    {
+      // Local copy: pack then immediately unpack (no MPI needed)
+      if ((length = Parallel::data_packer(0, ts->transfer_src[myrank], ts->transfer_dst[myrank],
+                                          node, PACK, ts->VarList1, ts->VarList2, ts->Symmetry)))
+      {
+        double *local_data = new double[length];
+        if (!local_data)
+        {
+          cout << "out of memory in transfer_begin, local copy" << endl;
+          MPI_Abort(MPI_COMM_WORLD, 1);
+        }
+        Parallel::data_packer(local_data, ts->transfer_src[myrank], ts->transfer_dst[myrank],
+                              node, PACK, ts->VarList1, ts->VarList2, ts->Symmetry);
+        Parallel::data_packer(local_data, ts->transfer_src[node], ts->transfer_dst[node],
+                              node, UNPACK, ts->VarList1, ts->VarList2, ts->Symmetry);
+        delete[] local_data;
+      }
+    }
+    else
+    {
+      // send from this cpu to cpu#node
+      if ((length = Parallel::data_packer(0, ts->transfer_src[myrank], ts->transfer_dst[myrank],
+                                          node, PACK, ts->VarList1, ts->VarList2, ts->Symmetry)))
+      {
+        ts->send_data[node] = new double[length];
+        if (!ts->send_data[node])
+        {
+          cout << "out of memory in transfer_begin, send" << endl;
+          MPI_Abort(MPI_COMM_WORLD, 1);
+        }
+        Parallel::data_packer(ts->send_data[node], ts->transfer_src[myrank], ts->transfer_dst[myrank],
+                              node, PACK, ts->VarList1, ts->VarList2, ts->Symmetry);
+        MPI_Isend((void *)ts->send_data[node], length, MPI_DOUBLE, node, 1, MPI_COMM_WORLD,
+                  ts->reqs + ts->req_no++);
+      }
+      // receive from cpu#node to this cpu
+      if ((length = Parallel::data_packer(0, ts->transfer_src[node], ts->transfer_dst[node],
+                                          node, UNPACK, ts->VarList1, ts->VarList2, ts->Symmetry)))
+      {
+        ts->rec_data[node] = new double[length];
+        if (!ts->rec_data[node])
+        {
+          cout << "out of memory in transfer_begin, recv" << endl;
+          MPI_Abort(MPI_COMM_WORLD, 1);
+        }
+        MPI_Irecv((void *)ts->rec_data[node], length, MPI_DOUBLE, node, 1, MPI_COMM_WORLD,
+                  ts->reqs + ts->req_no++);
+      }
+    }
+  }
+  // NOTE: MPI_Waitall is NOT called here - that happens in transfer_end
+}
+//
+static void transfer_end(Parallel::TransferState *ts)
+{
+  // Wait for all pending MPI operations
+  MPI_Waitall(ts->req_no, ts->reqs, ts->stats);
+
+  // Unpack received data from remote ranks
+  for (int node = 0; node < ts->cpusize; node++)
+    if (ts->rec_data[node])
+      Parallel::data_packer(ts->rec_data[node], ts->transfer_src[node], ts->transfer_dst[node],
+                            node, UNPACK, ts->VarList1, ts->VarList2, ts->Symmetry);
+
+  // Cleanup MPI buffers
+  for (int node = 0; node < ts->cpusize; node++)
+  {
+    if (ts->send_data[node])
+      delete[] ts->send_data[node];
+    if (ts->rec_data[node])
+      delete[] ts->rec_data[node];
+  }
+  delete[] ts->reqs;
+  delete[] ts->stats;
+  delete[] ts->send_data;
+  delete[] ts->rec_data;
+}
+//
+Parallel::SyncHandle *Parallel::SyncBegin(Patch *Pat, MyList<var> *VarList, int Symmetry)
+{
+  int cpusize;
+  MPI_Comm_size(MPI_COMM_WORLD, &cpusize);
+
+  SyncHandle *handle = new SyncHandle;
+  handle->num_states = 1;
+  handle->states = new TransferState[1];
+
+  TransferState *ts = &handle->states[0];
+  ts->cpusize = cpusize;
+  ts->VarList1 = VarList;
+  ts->VarList2 = VarList;
+  ts->Symmetry = Symmetry;
+  ts->owns_gsl = true;
+
+  ts->dst = build_ghost_gsl(Pat);
+  ts->src = new MyList<Parallel::gridseg> *[cpusize];
+  ts->transfer_src = new MyList<Parallel::gridseg> *[cpusize];
+  ts->transfer_dst = new MyList<Parallel::gridseg> *[cpusize];
+  for (int node = 0; node < cpusize; node++)
+  {
+    ts->src[node] = build_owned_gsl0(Pat, node);
+    build_gstl(ts->src[node], ts->dst, &ts->transfer_src[node], &ts->transfer_dst[node]);
+  }
+
+  transfer_begin(ts);
+
+  return handle;
+}
+//
+Parallel::SyncHandle *Parallel::SyncBegin(MyList<Patch> *PatL, MyList<var> *VarList, int Symmetry)
+{
+  int cpusize;
+  MPI_Comm_size(MPI_COMM_WORLD, &cpusize);
+
+  // Count patches
+  int num_patches = 0;
+  MyList<Patch> *Pp = PatL;
+  while (Pp) { num_patches++; Pp = Pp->next; }
+
+  SyncHandle *handle = new SyncHandle;
+  handle->num_states = num_patches + 1; // intra-patch transfers + 1 inter-patch transfer
+  handle->states = new TransferState[handle->num_states];
+
+  // Intra-patch sync: for each patch, build ghost lists and initiate transfer
+  int idx = 0;
+  Pp = PatL;
+  while (Pp)
+  {
+    TransferState *ts = &handle->states[idx];
+    ts->cpusize = cpusize;
+    ts->VarList1 = VarList;
+    ts->VarList2 = VarList;
+    ts->Symmetry = Symmetry;
+    ts->owns_gsl = true;
+
+    ts->dst = build_ghost_gsl(Pp->data);
+    ts->src = new MyList<Parallel::gridseg> *[cpusize];
+    ts->transfer_src = new MyList<Parallel::gridseg> *[cpusize];
+    ts->transfer_dst = new MyList<Parallel::gridseg> *[cpusize];
+    for (int node = 0; node < cpusize; node++)
+    {
+      ts->src[node] = build_owned_gsl0(Pp->data, node);
+      build_gstl(ts->src[node], ts->dst, &ts->transfer_src[node], &ts->transfer_dst[node]);
+    }
+
+    transfer_begin(ts);
+
+    idx++;
+    Pp = Pp->next;
+  }
+
+  // Inter-patch sync: buffer zone exchange between patches
+  {
+    TransferState *ts = &handle->states[idx];
+    ts->cpusize = cpusize;
+    ts->VarList1 = VarList;
+    ts->VarList2 = VarList;
+    ts->Symmetry = Symmetry;
+    ts->owns_gsl = true;
+
+    ts->dst = build_buffer_gsl(PatL);
+    ts->src = new MyList<Parallel::gridseg> *[cpusize];
+    ts->transfer_src = new MyList<Parallel::gridseg> *[cpusize];
+    ts->transfer_dst = new MyList<Parallel::gridseg> *[cpusize];
+    for (int node = 0; node < cpusize; node++)
+    {
+      ts->src[node] = build_owned_gsl(PatL, node, 5, Symmetry);
+      build_gstl(ts->src[node], ts->dst, &ts->transfer_src[node], &ts->transfer_dst[node]);
+    }
+
+    transfer_begin(ts);
+  }
+
+  return handle;
+}
+//
+void Parallel::SyncEnd(SyncHandle *handle)
+{
+  if (!handle)
+    return;
+
+  // Wait for all pending transfers and unpack
+  for (int i = 0; i < handle->num_states; i++)
+  {
+    TransferState *ts = &handle->states[i];
+    transfer_end(ts);
+
+    // Cleanup grid segment lists only if this state owns them
+    if (ts->owns_gsl)
+    {
+      if (ts->dst)
+        ts->dst->destroyList();
+      for (int node = 0; node < ts->cpusize; node++)
+      {
+        if (ts->src[node])
+          ts->src[node]->destroyList();
+        if (ts->transfer_src[node])
+          ts->transfer_src[node]->destroyList();
+        if (ts->transfer_dst[node])
+          ts->transfer_dst[node]->destroyList();
+      }
+      delete[] ts->src;
+      delete[] ts->transfer_src;
+      delete[] ts->transfer_dst;
+    }
+  }
+
+  delete[] handle->states;
+  delete handle;
+}
+//
+// SyncPreparePlan: Pre-build grid segment lists for a patch list.
+// The plan can be reused across multiple SyncBeginWithPlan calls
+// as long as the mesh topology does not change (no regridding).
+//
+Parallel::SyncPlan *Parallel::SyncPreparePlan(MyList<Patch> *PatL, int Symmetry)
+{
+  int cpusize;
+  MPI_Comm_size(MPI_COMM_WORLD, &cpusize);
+
+  // Count patches
+  int num_patches = 0;
+  MyList<Patch> *Pp = PatL;
+  while (Pp) { num_patches++; Pp = Pp->next; }
+
+  SyncPlan *plan = new SyncPlan;
+  plan->num_entries = num_patches + 1; // intra-patch + 1 inter-patch
+  plan->Symmetry = Symmetry;
+  plan->entries = new SyncPlanEntry[plan->num_entries];
+
+  // Intra-patch entries: ghost zone exchange within each patch
+  int idx = 0;
+  Pp = PatL;
+  while (Pp)
+  {
+    SyncPlanEntry *pe = &plan->entries[idx];
+    pe->cpusize = cpusize;
+    pe->dst = build_ghost_gsl(Pp->data);
+    pe->src = new MyList<Parallel::gridseg> *[cpusize];
+    pe->transfer_src = new MyList<Parallel::gridseg> *[cpusize];
+    pe->transfer_dst = new MyList<Parallel::gridseg> *[cpusize];
+    for (int node = 0; node < cpusize; node++)
+    {
+      pe->src[node] = build_owned_gsl0(Pp->data, node);
+      build_gstl(pe->src[node], pe->dst, &pe->transfer_src[node], &pe->transfer_dst[node]);
+    }
+    idx++;
+    Pp = Pp->next;
+  }
+
+  // Inter-patch entry: buffer zone exchange between patches
+  {
+    SyncPlanEntry *pe = &plan->entries[idx];
+    pe->cpusize = cpusize;
+    pe->dst = build_buffer_gsl(PatL);
+    pe->src = new MyList<Parallel::gridseg> *[cpusize];
+    pe->transfer_src = new MyList<Parallel::gridseg> *[cpusize];
+    pe->transfer_dst = new MyList<Parallel::gridseg> *[cpusize];
+    for (int node = 0; node < cpusize; node++)
+    {
+      pe->src[node] = build_owned_gsl(PatL, node, 5, Symmetry);
+      build_gstl(pe->src[node], pe->dst, &pe->transfer_src[node], &pe->transfer_dst[node]);
+    }
+  }
+
+  return plan;
+}
+//
+void Parallel::SyncFreePlan(SyncPlan *plan)
+{
+  if (!plan)
+    return;
+
+  for (int i = 0; i < plan->num_entries; i++)
+  {
+    SyncPlanEntry *pe = &plan->entries[i];
+    if (pe->dst)
+      pe->dst->destroyList();
+    for (int node = 0; node < pe->cpusize; node++)
+    {
+      if (pe->src[node])
+        pe->src[node]->destroyList();
+      if (pe->transfer_src[node])
+        pe->transfer_src[node]->destroyList();
+      if (pe->transfer_dst[node])
+        pe->transfer_dst[node]->destroyList();
+    }
+    delete[] pe->src;
+    delete[] pe->transfer_src;
+    delete[] pe->transfer_dst;
+  }
+  delete[] plan->entries;
+  delete plan;
+}
+//
+// SyncBeginWithPlan: Use pre-built GSLs from a SyncPlan to initiate async transfer.
+// This avoids the O(cpusize * blocks^2) cost of rebuilding GSLs on every call.
+//
+Parallel::SyncHandle *Parallel::SyncBeginWithPlan(SyncPlan *plan, MyList<var> *VarList)
+{
+  return SyncBeginWithPlan(plan, VarList, VarList);
+}
+//
+Parallel::SyncHandle *Parallel::SyncBeginWithPlan(SyncPlan *plan, MyList<var> *VarList1, MyList<var> *VarList2)
+{
+  SyncHandle *handle = new SyncHandle;
+  handle->num_states = plan->num_entries;
+  handle->states = new TransferState[handle->num_states];
+
+  for (int i = 0; i < plan->num_entries; i++)
+  {
+    SyncPlanEntry *pe = &plan->entries[i];
+    TransferState *ts = &handle->states[i];
+
+    ts->cpusize = pe->cpusize;
+    ts->VarList1 = VarList1;
+    ts->VarList2 = VarList2;
+    ts->Symmetry = plan->Symmetry;
+    ts->owns_gsl = false; // GSLs are owned by the plan, not this handle
+
+    // Borrow GSL pointers from the plan (do NOT free them in SyncEnd)
+    ts->transfer_src = pe->transfer_src;
+    ts->transfer_dst = pe->transfer_dst;
+    ts->src = pe->src;
+    ts->dst = pe->dst;
+
+    transfer_begin(ts);
+  }
+
+  return handle;
+}
 // collect buffer grid segments or blocks for the periodic boundary condition of given patch
 // ---------------------------------------------------
 // |con |                                       |con |
--- a/AMSS_NCKU_source/Parallel.h
+++ b/AMSS_NCKU_source/Parallel.h
@@ -81,6 +81,53 @@ namespace Parallel
                   int Symmetry);
  void Sync(Patch *Pat, MyList<var> *VarList, int Symmetry);
  void Sync(MyList<Patch> *PatL, MyList<var> *VarList, int Symmetry);
+
+  // Async Sync: overlap MPI communication with computation
+  struct TransferState
+  {
+    MPI_Request *reqs;
+    MPI_Status *stats;
+    int req_no;
+    double **send_data;
+    double **rec_data;
+    int cpusize;
+    MyList<gridseg> **transfer_src;
+    MyList<gridseg> **transfer_dst;
+    MyList<gridseg> **src;
+    MyList<gridseg> *dst;
+    MyList<var> *VarList1;
+    MyList<var> *VarList2;
+    int Symmetry;
+    bool owns_gsl; // true if this state owns and should free the GSLs
+  };
+  struct SyncHandle
+  {
+    TransferState *states;
+    int num_states;
+  };
+  SyncHandle *SyncBegin(Patch *Pat, MyList<var> *VarList, int Symmetry);
+  SyncHandle *SyncBegin(MyList<Patch> *PatL, MyList<var> *VarList, int Symmetry);
+  void SyncEnd(SyncHandle *handle);
+
+  // Cached GSL plan: pre-build grid segment lists once, reuse across multiple Sync calls
+  struct SyncPlanEntry
+  {
+    int cpusize;
+    MyList<gridseg> **transfer_src;
+    MyList<gridseg> **transfer_dst;
+    MyList<gridseg> **src;
+    MyList<gridseg> *dst;
+  };
+  struct SyncPlan
+  {
+    SyncPlanEntry *entries;
+    int num_entries;
+    int Symmetry;
+  };
+  SyncPlan *SyncPreparePlan(MyList<Patch> *PatL, int Symmetry);
+  void SyncFreePlan(SyncPlan *plan);
+  SyncHandle *SyncBeginWithPlan(SyncPlan *plan, MyList<var> *VarList);
+  SyncHandle *SyncBeginWithPlan(SyncPlan *plan, MyList<var> *VarList1, MyList<var> *VarList2);
  void OutBdLow2Hi(Patch *Patc, Patch *Patf,
                   MyList<var> *VarList1 /* source */, MyList<var> *VarList2 /* target */,
                   int Symmetry);
--- a/AMSS_NCKU_source/Z4c_class.C
+++ b/AMSS_NCKU_source/Z4c_class.C
@@ -186,6 +186,12 @@ void Z4c_class::Step(int lev, int YN)
  int ERROR = 0;

  MyList<ss_patch> *sPp;
+
+  // Pre-build grid segment lists once for this level's patches.
+  // These are reused across predictor + 3 corrector SyncBegin calls,
+  // avoiding O(cpusize * blocks^2) rebuild each time.
+  Parallel::SyncPlan *sync_plan = Parallel::SyncPreparePlan(GH->PatL[lev], Symmetry);
+
  // Predictor
  MyList<Patch> *Pp = GH->PatL[lev];
  while (Pp)
@@ -321,13 +327,17 @@ void Z4c_class::Step(int lev, int YN)
    }
    Pp = Pp->next;
  }
-  // check error information
+  // Start async ghost zone exchange - overlaps with error check and Shell computation
+  Parallel::SyncHandle *sync_pre = Parallel::SyncBeginWithPlan(sync_plan, SynchList_pre);
+
+  // check error information (overlaps with MPI transfer)
  {
    int erh = ERROR;
    MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
  }
  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    Parallel::Dump_Data(GH->PatL[lev], StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -475,6 +485,7 @@ void Z4c_class::Step(int lev, int YN)
  }
  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    SH->Dump_Data(StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -485,7 +496,8 @@ void Z4c_class::Step(int lev, int YN)
  }
 #endif

-  Parallel::Sync(GH->PatL[lev], SynchList_pre, Symmetry);
+  // Complete async ghost zone exchange
+  if (sync_pre) Parallel::SyncEnd(sync_pre);

 #ifdef WithShell
  if (lev == 0)
@@ -693,13 +705,17 @@ void Z4c_class::Step(int lev, int YN)
      Pp = Pp->next;
    }

-    // check error information
+    // Start async ghost zone exchange - overlaps with error check and Shell computation
+    Parallel::SyncHandle *sync_cor = Parallel::SyncBeginWithPlan(sync_plan, SynchList_cor);
+
+    // check error information (overlaps with MPI transfer)
    {
      int erh = ERROR;
      MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
    }
    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      Parallel::Dump_Data(GH->PatL[lev], SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -857,6 +873,7 @@ void Z4c_class::Step(int lev, int YN)
    }
    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      SH->Dump_Data(SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -868,7 +885,8 @@ void Z4c_class::Step(int lev, int YN)
    }
 #endif

-    Parallel::Sync(GH->PatL[lev], SynchList_cor, Symmetry);
+    // Complete async ghost zone exchange
+    if (sync_cor) Parallel::SyncEnd(sync_cor);

 #ifdef WithShell
    if (lev == 0)
@@ -1042,6 +1060,8 @@ void Z4c_class::Step(int lev, int YN)
      Porg0[ithBH][2] = Porg1[ithBH][2];
    }
  }
+
+  Parallel::SyncFreePlan(sync_plan);
 }
 #else
 // for constraint preserving boundary (CPBC)
@@ -1075,6 +1095,10 @@ void Z4c_class::Step(int lev, int YN)
  int ERROR = 0;

  MyList<ss_patch> *sPp;
+
+  // Pre-build grid segment lists once for this level's patches.
+  Parallel::SyncPlan *sync_plan = Parallel::SyncPreparePlan(GH->PatL[lev], Symmetry);
+
  // Predictor
  MyList<Patch> *Pp = GH->PatL[lev];
  while (Pp)
@@ -1542,13 +1566,17 @@ void Z4c_class::Step(int lev, int YN)
  }
 #endif
  }
-  // check error information
+  // Start async ghost zone exchange - overlaps with error check
+  Parallel::SyncHandle *sync_pre = Parallel::SyncBeginWithPlan(sync_plan, SynchList_pre);
+
+  // check error information (overlaps with MPI transfer)
  {
    int erh = ERROR;
    MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
  }
  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    SH->Dump_Data(StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -1558,7 +1586,8 @@ void Z4c_class::Step(int lev, int YN)
    }
  }

-  Parallel::Sync(GH->PatL[lev], SynchList_pre, Symmetry);
+  // Complete async ghost zone exchange
+  if (sync_pre) Parallel::SyncEnd(sync_pre);

  if (lev == 0)
  {
@@ -2103,13 +2132,17 @@ void Z4c_class::Step(int lev, int YN)
        sPp = sPp->next;
      }
    }
-    // check error information
+    // Start async ghost zone exchange - overlaps with error check
+    Parallel::SyncHandle *sync_cor = Parallel::SyncBeginWithPlan(sync_plan, SynchList_cor);
+
+    // check error information (overlaps with MPI transfer)
    {
      int erh = ERROR;
      MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
    }
    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      SH->Dump_Data(SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -2120,7 +2153,8 @@ void Z4c_class::Step(int lev, int YN)
      }
    }

-    Parallel::Sync(GH->PatL[lev], SynchList_cor, Symmetry);
+    // Complete async ghost zone exchange
+    if (sync_cor) Parallel::SyncEnd(sync_cor);

    if (lev == 0)
    {
@@ -2346,6 +2380,8 @@ void Z4c_class::Step(int lev, int YN)
 	  DG_List->clearList();
 	}
 #endif
+
+  Parallel::SyncFreePlan(sync_plan);
 }
 #endif
 #undef MRBD
--- a/AMSS_NCKU_source/bssn_class.C
+++ b/AMSS_NCKU_source/bssn_class.C
@@ -3035,6 +3035,12 @@ void bssn_class::Step(int lev, int YN)
  int ERROR = 0;

  MyList<ss_patch> *sPp;
+
+  // Pre-build grid segment lists once for this level's patches.
+  // These are reused across predictor + 3 corrector SyncBegin calls,
+  // avoiding O(cpusize * blocks^2) rebuild each time.
+  Parallel::SyncPlan *sync_plan = Parallel::SyncPreparePlan(GH->PatL[lev], Symmetry);
+
  // Predictor
  MyList<Patch> *Pp = GH->PatL[lev];
  while (Pp)
@@ -3158,13 +3164,18 @@ void bssn_class::Step(int lev, int YN)
    }
    Pp = Pp->next;
  }
-  // check error information
+
+  // Start async ghost zone exchange - overlaps with error check and Shell computation
+  Parallel::SyncHandle *sync_pre = Parallel::SyncBeginWithPlan(sync_plan, SynchList_pre);
+
+  // check error information (overlaps with MPI transfer)
  {
    int erh = ERROR;
    MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
  }
  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    Parallel::Dump_Data(GH->PatL[lev], StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -3324,6 +3335,7 @@ void bssn_class::Step(int lev, int YN)

  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    SH->Dump_Data(StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -3334,7 +3346,8 @@ void bssn_class::Step(int lev, int YN)
  }
 #endif

-  Parallel::Sync(GH->PatL[lev], SynchList_pre, Symmetry);
+  // Complete async ghost zone exchange
+  if (sync_pre) Parallel::SyncEnd(sync_pre);

 #ifdef WithShell
  if (lev == 0)
@@ -3528,7 +3541,10 @@ void bssn_class::Step(int lev, int YN)
      Pp = Pp->next;
    }

-    // check error information
+    // Start async ghost zone exchange - overlaps with error check and Shell computation
+    Parallel::SyncHandle *sync_cor = Parallel::SyncBeginWithPlan(sync_plan, SynchList_cor);
+
+    // check error information (overlaps with MPI transfer)
    {
      int erh = ERROR;
      MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
@@ -3536,6 +3552,7 @@ void bssn_class::Step(int lev, int YN)

    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      Parallel::Dump_Data(GH->PatL[lev], SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -3692,6 +3709,7 @@ void bssn_class::Step(int lev, int YN)
    }
    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      SH->Dump_Data(SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -3704,7 +3722,8 @@ void bssn_class::Step(int lev, int YN)
    }
 #endif

-    Parallel::Sync(GH->PatL[lev], SynchList_cor, Symmetry);
+    // Complete async ghost zone exchange
+    if (sync_cor) Parallel::SyncEnd(sync_cor);

 #ifdef WithShell
    if (lev == 0)
@@ -3895,6 +3914,8 @@ void bssn_class::Step(int lev, int YN)
      Porg0[ithBH][2] = Porg1[ithBH][2];
    }
  }
+
+  Parallel::SyncFreePlan(sync_plan);
 }

 //================================================================================================
@@ -4817,6 +4838,12 @@ void bssn_class::Step(int lev, int YN)
  int ERROR = 0;

  MyList<ss_patch> *sPp;
+
+  // Pre-build grid segment lists once for this level's patches.
+  // These are reused across predictor + 3 corrector SyncBegin calls,
+  // avoiding O(cpusize * blocks^2) rebuild each time.
+  Parallel::SyncPlan *sync_plan = Parallel::SyncPreparePlan(GH->PatL[lev], Symmetry);
+
  // Predictor
  MyList<Patch> *Pp = GH->PatL[lev];
  while (Pp)
@@ -4943,13 +4970,17 @@ void bssn_class::Step(int lev, int YN)

  //   misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"after Predictor rhs calculation");

-  // check error information
+  // Start async ghost zone exchange - overlaps with error check and BH position
+  Parallel::SyncHandle *sync_pre = Parallel::SyncBeginWithPlan(sync_plan, SynchList_pre);
+
+  // check error information (overlaps with MPI transfer)
  {
    int erh = ERROR;
    MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, GH->Commlev[lev]);
  }
  if (ERROR)
  {
+    Parallel::SyncEnd(sync_pre); sync_pre = 0;
    Parallel::Dump_Data(GH->PatL[lev], StateList, 0, PhysTime, dT_lev);
    if (myrank == 0)
    {
@@ -4961,7 +4992,8 @@ void bssn_class::Step(int lev, int YN)

  //   misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"before Predictor sync");

-  Parallel::Sync(GH->PatL[lev], SynchList_pre, Symmetry);
+  // Complete async ghost zone exchange
+  if (sync_pre) Parallel::SyncEnd(sync_pre);

 #if (MAPBH == 0)
  // for black hole position
@@ -5140,13 +5172,17 @@ void bssn_class::Step(int lev, int YN)

    //   misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"before Corrector error check");

-    // check error information
+    // Start async ghost zone exchange - overlaps with error check and BH position
+    Parallel::SyncHandle *sync_cor = Parallel::SyncBeginWithPlan(sync_plan, SynchList_cor);
+
+    // check error information (overlaps with MPI transfer)
    {
      int erh = ERROR;
      MPI_Allreduce(&erh, &ERROR, 1, MPI_INT, MPI_SUM, GH->Commlev[lev]);
    }
    if (ERROR)
    {
+      Parallel::SyncEnd(sync_cor); sync_cor = 0;
      Parallel::Dump_Data(GH->PatL[lev], SynchList_pre, 0, PhysTime, dT_lev);
      if (myrank == 0)
      {
@@ -5160,7 +5196,8 @@ void bssn_class::Step(int lev, int YN)

    //    misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"before Corrector sync");

-    Parallel::Sync(GH->PatL[lev], SynchList_cor, Symmetry);
+    // Complete async ghost zone exchange
+    if (sync_cor) Parallel::SyncEnd(sync_cor);

    //    misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"after Corrector sync");

@@ -5276,6 +5313,8 @@ void bssn_class::Step(int lev, int YN)

  //     if(myrank==GH->start_rank[lev]) cout<<GH->mylev<<endl;
  //     misc::tillherecheck(GH->Commlev[lev],GH->start_rank[lev],"complet GH Step");
+
+  Parallel::SyncFreePlan(sync_plan);
 }

 //================================================================================================
--- a/AMSS_NCKU_source/bssn_rhs.f90
+++ b/AMSS_NCKU_source/bssn_rhs.f90
@@ -161,8 +161,36 @@

  chi_rhs = F2o3 *chin1*( alpn1 * trK - div_beta ) !rhs for chi

+  call fderivs(ex,dxx,gxxx,gxxy,gxxz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)
+  call fderivs(ex,gxy,gxyx,gxyy,gxyz,X,Y,Z,ANTI,ANTI,SYM ,Symmetry,Lev)
+  call fderivs(ex,gxz,gxzx,gxzy,gxzz,X,Y,Z,ANTI,SYM ,ANTI,Symmetry,Lev)
+  call fderivs(ex,dyy,gyyx,gyyy,gyyz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)
+  call fderivs(ex,gyz,gyzx,gyzy,gyzz,X,Y,Z,SYM ,ANTI,ANTI,Symmetry,Lev)
+  call fderivs(ex,dzz,gzzx,gzzy,gzzz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)

+  gxx_rhs = - TWO * alpn1 * Axx    -  F2o3 * gxx * div_beta          + &
+              TWO *(  gxx * betaxx +   gxy * betayx +   gxz * betazx)

+  gyy_rhs = - TWO * alpn1 * Ayy    -  F2o3 * gyy * div_beta          + &
+              TWO *(  gxy * betaxy +   gyy * betayy +   gyz * betazy)
+
+  gzz_rhs = - TWO * alpn1 * Azz    -  F2o3 * gzz * div_beta          + &
+              TWO *(  gxz * betaxz +   gyz * betayz +   gzz * betazz)
+
+  gxy_rhs = - TWO * alpn1 * Axy    +  F1o3 * gxy    * div_beta       + &
+                      gxx * betaxy                  +   gxz * betazy + &
+                                       gyy * betayx +   gyz * betazx   &
+                                                    -   gxy * betazz
+
+  gyz_rhs = - TWO * alpn1 * Ayz    +  F1o3 * gyz    * div_beta       + &
+                      gxy * betaxz +   gyy * betayz                  + &
+                      gxz * betaxy                  +   gzz * betazy   &
+                                                    -   gyz * betaxx
+ 
+  gxz_rhs = - TWO * alpn1 * Axz    +  F1o3 * gxz    * div_beta       + &
+                      gxx * betaxz +   gxy * betayz                  + &
+                                       gyz * betayx +   gzz * betazx   &
+                                                    -   gxz * betayy     !rhs for gij

 ! invert tilted metric
  gupzz =  gxx * gyy * gzz + gxy * gyz * gxz + gxz * gxy * gyz - &
@@ -173,12 +201,7 @@
  gupyy =   ( gxx * gzz - gxz * gxz ) / gupzz
  gupyz = - ( gxx * gyz - gxy * gxz ) / gupzz
  gupzz =   ( gxx * gyy - gxy * gxy ) / gupzz
-  call fderivs(ex,dxx,gxxx,gxxy,gxxz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)
-  call fderivs(ex,gxy,gxyx,gxyy,gxyz,X,Y,Z,ANTI,ANTI,SYM ,Symmetry,Lev)
-  call fderivs(ex,gxz,gxzx,gxzy,gxzz,X,Y,Z,ANTI,SYM ,ANTI,Symmetry,Lev)
-  call fderivs(ex,dyy,gyyx,gyyy,gyyz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)
-  call fderivs(ex,gyz,gyzx,gyzy,gyzz,X,Y,Z,SYM ,ANTI,ANTI,Symmetry,Lev)
-  call fderivs(ex,dzz,gzzx,gzzy,gzzz,X,Y,Z,SYM ,SYM ,SYM ,Symmetry,Lev)
+
  if(co == 0)then
 ! Gam^i_Res = Gam^i + gup^ij_,j
  Gmx_Res = Gamx - (gupxx*(gupxx*gxxx+gupxy*gxyx+gupxz*gxzx)&
@@ -922,103 +945,60 @@
  SSA(2)=SYM
  SSA(3)=ANTI

-!!!!!!!!!advection term part
+!!!!!!!!!advection term + Kreiss-Oliger dissipation (merged for cache efficiency)
+! lopsided_kodis shares the symmetry_bd buffer between advection and
+! dissipation, eliminating redundant full-grid copies. For metric variables
+! gxx/gyy/gzz (=dxx/dyy/dzz+1): kodis stencil coefficients sum to zero,
+! so the constant offset has no effect on dissipation.

-  gxx_rhs = - TWO * alpn1 * Axx    -  F2o3 * gxx * div_beta          + &
-              TWO *(  gxx * betaxx +   gxy * betayx +   gxz * betazx)
+  call lopsided_kodis(ex,X,Y,Z,gxx,gxx_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+  call lopsided_kodis(ex,X,Y,Z,gxy,gxy_rhs,betax,betay,betaz,Symmetry,AAS,eps)
+  call lopsided_kodis(ex,X,Y,Z,gxz,gxz_rhs,betax,betay,betaz,Symmetry,ASA,eps)
+  call lopsided_kodis(ex,X,Y,Z,gyy,gyy_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+  call lopsided_kodis(ex,X,Y,Z,gyz,gyz_rhs,betax,betay,betaz,Symmetry,SAA,eps)
+  call lopsided_kodis(ex,X,Y,Z,gzz,gzz_rhs,betax,betay,betaz,Symmetry,SSS,eps)

-  gyy_rhs = - TWO * alpn1 * Ayy    -  F2o3 * gyy * div_beta          + &
-              TWO *(  gxy * betaxy +   gyy * betayy +   gyz * betazy)
+  call lopsided_kodis(ex,X,Y,Z,Axx,Axx_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+  call lopsided_kodis(ex,X,Y,Z,Axy,Axy_rhs,betax,betay,betaz,Symmetry,AAS,eps)
+  call lopsided_kodis(ex,X,Y,Z,Axz,Axz_rhs,betax,betay,betaz,Symmetry,ASA,eps)
+  call lopsided_kodis(ex,X,Y,Z,Ayy,Ayy_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+  call lopsided_kodis(ex,X,Y,Z,Ayz,Ayz_rhs,betax,betay,betaz,Symmetry,SAA,eps)
+  call lopsided_kodis(ex,X,Y,Z,Azz,Azz_rhs,betax,betay,betaz,Symmetry,SSS,eps)

-  gzz_rhs = - TWO * alpn1 * Azz    -  F2o3 * gzz * div_beta          + &
-              TWO *(  gxz * betaxz +   gyz * betayz +   gzz * betazz)
+  call lopsided_kodis(ex,X,Y,Z,chi,chi_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+  call lopsided_kodis(ex,X,Y,Z,trK,trK_rhs,betax,betay,betaz,Symmetry,SSS,eps)

-  gxy_rhs = - TWO * alpn1 * Axy    +  F1o3 * gxy    * div_beta       + &
-                      gxx * betaxy                  +   gxz * betazy + &
-                                        gyy * betayx +   gyz * betazx   &
-                                                    -   gxy * betazz
+  call lopsided_kodis(ex,X,Y,Z,Gamx,Gamx_rhs,betax,betay,betaz,Symmetry,ASS,eps)
+  call lopsided_kodis(ex,X,Y,Z,Gamy,Gamy_rhs,betax,betay,betaz,Symmetry,SAS,eps)
+  call lopsided_kodis(ex,X,Y,Z,Gamz,Gamz_rhs,betax,betay,betaz,Symmetry,SSA,eps)

-  gyz_rhs = - TWO * alpn1 * Ayz    +  F1o3 * gyz    * div_beta       + &
-                      gxy * betaxz +   gyy * betayz                  + &
-                      gxz * betaxy                  +   gzz * betazy   &
-                                                    -   gyz * betaxx
-
-  gxz_rhs = - TWO * alpn1 * Axz    +  F1o3 * gxz    * div_beta       + &
-                      gxx * betaxz +   gxy * betayz                  + &
-                                        gyz * betayx +   gzz * betazx   &
-                                                    -   gxz * betayy     !rhs for gij
-
-
-
-
-
-  if(eps>0)then 
-! usual Kreiss-Oliger dissipation     
-  call merge_lopsided_kodis(ex,X,Y,Z,chi,chi_rhs,betax,betay,betaz,Symmetry,SSS,eps) 
-  call merge_lopsided_kodis(ex,X,Y,Z,gxx,gxx_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,gxy,gxy_rhs,betax,betay,betaz,Symmetry,AAS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,gxz,gxz_rhs,betax,betay,betaz,Symmetry,ASA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,gyy,gyy_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,gyz,gyz_rhs,betax,betay,betaz,Symmetry,SAA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,gzz,gzz_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Axx,Axx_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Axy,Axy_rhs,betax,betay,betaz,Symmetry,AAS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Axz,Axz_rhs,betax,betay,betaz,Symmetry,ASA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Ayy,Ayy_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Ayz,Ayz_rhs,betax,betay,betaz,Symmetry,SAA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Azz,Azz_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,chi,chi_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,trK,trK_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Gamx,Gamx_rhs,betax,betay,betaz,Symmetry,ASS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Gamy,Gamy_rhs,betax,betay,betaz,Symmetry,SAS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Gamz,Gamz_rhs,betax,betay,betaz,Symmetry,SSA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,Lap,Lap_rhs,betax,betay,betaz,Symmetry,SSS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,betax,betax_rhs,betax,betay,betaz,Symmetry,ASS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,betay,betay_rhs,betax,betay,betaz,Symmetry,SAS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,betaz,betaz_rhs,betax,betay,betaz,Symmetry,SSA,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,dtSfx,dtSfx_rhs,betax,betay,betaz,Symmetry,ASS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,dtSfy,dtSfy_rhs,betax,betay,betaz,Symmetry,SAS,eps)
-  call merge_lopsided_kodis(ex,X,Y,Z,dtSfz,dtSfz_rhs,betax,betay,betaz,Symmetry,SSA,eps)
-
-
-
-
-
-
-
-
-
-
-
-
-  else 
-  call lopsided(ex,X,Y,Z,gxx,gxx_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,gxy,gxy_rhs,betax,betay,betaz,Symmetry,AAS)
-  call lopsided(ex,X,Y,Z,gxz,gxz_rhs,betax,betay,betaz,Symmetry,ASA)
-  call lopsided(ex,X,Y,Z,gyy,gyy_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,gyz,gyz_rhs,betax,betay,betaz,Symmetry,SAA)
-  call lopsided(ex,X,Y,Z,gzz,gzz_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,Axx,Axx_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,Axy,Axy_rhs,betax,betay,betaz,Symmetry,AAS)
-  call lopsided(ex,X,Y,Z,Axz,Axz_rhs,betax,betay,betaz,Symmetry,ASA)
-  call lopsided(ex,X,Y,Z,Ayy,Ayy_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,Ayz,Ayz_rhs,betax,betay,betaz,Symmetry,SAA)
-  call lopsided(ex,X,Y,Z,Azz,Azz_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,chi,chi_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,trK,trK_rhs,betax,betay,betaz,Symmetry,SSS)
-  call lopsided(ex,X,Y,Z,Gamx,Gamx_rhs,betax,betay,betaz,Symmetry,ASS)
-  call lopsided(ex,X,Y,Z,Gamy,Gamy_rhs,betax,betay,betaz,Symmetry,SAS)
-  call lopsided(ex,X,Y,Z,Gamz,Gamz_rhs,betax,betay,betaz,Symmetry,SSA)
+#if 1 
+!! bam does not apply dissipation on gauge variables
+  call lopsided_kodis(ex,X,Y,Z,Lap,Lap_rhs,betax,betay,betaz,Symmetry,SSS,eps)
+#if (GAUGE == 0 || GAUGE == 1 || GAUGE == 2 || GAUGE == 3 || GAUGE == 4 || GAUGE == 5 || GAUGE == 6 || GAUGE == 7)
+  call lopsided_kodis(ex,X,Y,Z,betax,betax_rhs,betax,betay,betaz,Symmetry,ASS,eps)
+  call lopsided_kodis(ex,X,Y,Z,betay,betay_rhs,betax,betay,betaz,Symmetry,SAS,eps)
+  call lopsided_kodis(ex,X,Y,Z,betaz,betaz_rhs,betax,betay,betaz,Symmetry,SSA,eps)
+#endif
+#if (GAUGE == 0 || GAUGE == 2 || GAUGE == 3 || GAUGE == 6 || GAUGE == 7)
+  call lopsided_kodis(ex,X,Y,Z,dtSfx,dtSfx_rhs,betax,betay,betaz,Symmetry,ASS,eps)
+  call lopsided_kodis(ex,X,Y,Z,dtSfy,dtSfy_rhs,betax,betay,betaz,Symmetry,SAS,eps)
+  call lopsided_kodis(ex,X,Y,Z,dtSfz,dtSfz_rhs,betax,betay,betaz,Symmetry,SSA,eps)
+#endif
+#else
+! No dissipation on gauge variables (advection only)
  call lopsided(ex,X,Y,Z,Lap,Lap_rhs,betax,betay,betaz,Symmetry,SSS)
+#if (GAUGE == 0 || GAUGE == 1 || GAUGE == 2 || GAUGE == 3 || GAUGE == 4 || GAUGE == 5 || GAUGE == 6 || GAUGE == 7)
  call lopsided(ex,X,Y,Z,betax,betax_rhs,betax,betay,betaz,Symmetry,ASS)
  call lopsided(ex,X,Y,Z,betay,betay_rhs,betax,betay,betaz,Symmetry,SAS)
  call lopsided(ex,X,Y,Z,betaz,betaz_rhs,betax,betay,betaz,Symmetry,SSA)
+#endif
+#if (GAUGE == 0 || GAUGE == 2 || GAUGE == 3 || GAUGE == 6 || GAUGE == 7)
  call lopsided(ex,X,Y,Z,dtSfx,dtSfx_rhs,betax,betay,betaz,Symmetry,ASS)
  call lopsided(ex,X,Y,Z,dtSfy,dtSfy_rhs,betax,betay,betaz,Symmetry,SAS)
  call lopsided(ex,X,Y,Z,dtSfz,dtSfz_rhs,betax,betay,betaz,Symmetry,SSA)
-
-
-  endif
+#endif
+#endif

  if(co == 0)then
 ! ham_Res = trR + 2/3 * K^2 - A_ij * A^ij - 16 * PI * rho
@@ -1163,265 +1143,3 @@ endif
  return

  end function compute_rhs_bssn
-
-
-
-
-  subroutine merge_lopsided_kodis(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA,eps)
-    implicit none
-
-  !~~~~~~> Input parameters:
-
-    integer, intent(in)  :: ex(1:3),Symmetry
-    real*8,  intent(in)  :: X(1:ex(1)),Y(1:ex(2)),Z(1:ex(3))
-    real*8,dimension(ex(1),ex(2),ex(3)),intent(in)   :: f,Sfx,Sfy,Sfz
-
-    real*8,dimension(ex(1),ex(2),ex(3)),intent(inout):: f_rhs
-    real*8,dimension(3),intent(in) ::SoA
-
-  !~~~~~~> local variables:
-  ! note index -2,-1,0, so we have 3 extra points
-    real*8,dimension(-2:ex(1),-2:ex(2),-2:ex(3))   :: fh
-    integer :: imin_lopsided,jmin_lopsided,kmin_lopsided,imin_kodis,jmin_kodis,kmin_kodis,imax,jmax,kmax,i,j,k
-    real*8 :: dX,dY,dZ
-    real*8 :: d12dx,d12dy,d12dz,d2dx,d2dy,d2dz
-    real*8,  parameter :: ZEO=0.d0,ONE=1.d0, F3=3.d0
-    real*8,  parameter :: TWO=2.d0,F6=6.0d0,F18=1.8d1
-    real*8,  parameter :: F12=1.2d1, F10=1.d1,EIT=8.d0
-    integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
-    real*8, parameter :: SIX=6.d0,FIT=1.5d1,TWT=2.d1
-    real*8,parameter::cof=6.4d1   ! 2^6
-    real*8,intent(in) :: eps
-    dX = X(2)-X(1)
-    dY = Y(2)-Y(1)
-    dZ = Z(2)-Z(1)
-
-    d12dx = ONE/F12/dX
-    d12dy = ONE/F12/dY
-    d12dz = ONE/F12/dZ
-
-    d2dx = ONE/TWO/dX
-    d2dy = ONE/TWO/dY
-    d2dz = ONE/TWO/dZ
-
-    imax = ex(1)
-    jmax = ex(2)
-    kmax = ex(3)
-
-    imin_lopsided = 1
-    jmin_lopsided = 1
-    kmin_lopsided = 1
-    if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin_lopsided = -2
-    if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin_lopsided = -2
-    if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin_lopsided = -2
-
-    imin_kodis = 1
-    jmin_kodis = 1
-    kmin_kodis = 1
-
-    if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin_kodis = -2
-    if(Symmetry == OCTANT .and. dabs(X(1)) < dX) imin_kodis = -2
-    if(Symmetry == OCTANT .and. dabs(Y(1)) < dY) jmin_kodis = -2
-
-
-    call symmetry_bd(3,ex,f,fh,SoA)
-
-  ! upper bound set ex-1 only for efficiency, 
-  ! the loop body will set ex 0 also
-    do k=1,ex(3)-1
-    do j=1,ex(2)-1
-    do i=1,ex(1)-1
-
-  !! new code, 2012dec27, based on bam
-  ! x direction   
-      if(Sfx(i,j,k) > ZEO)then
-        if(i+3 <= imax)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfx(i,j,k)*d12dx*(-F3*fh(i-1,j,k)-F10*fh(i,j,k)+F18*fh(i+1,j,k) &
-                                      -F6*fh(i+2,j,k)+    fh(i+3,j,k))
-      elseif(i+2 <= imax)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
-
-      elseif(i+1 <= imax)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfx(i,j,k)*d12dx*(-F3*fh(i+1,j,k)-F10*fh(i,j,k)+F18*fh(i-1,j,k) &
-                                      -F6*fh(i-2,j,k)+    fh(i-3,j,k))
-  ! set imax and imin_lopsided 0
-      endif
-    elseif(Sfx(i,j,k) < ZEO)then
-        if(i-3 >= imin_lopsided)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfx(i,j,k)*d12dx*(-F3*fh(i+1,j,k)-F10*fh(i,j,k)+F18*fh(i-1,j,k) &
-                                      -F6*fh(i-2,j,k)+    fh(i-3,j,k))
-      elseif(i-2 >= imin_lopsided)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
-
-      elseif(i-1 >= imin_lopsided)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfx(i,j,k)*d12dx*(-F3*fh(i-1,j,k)-F10*fh(i,j,k)+F18*fh(i+1,j,k) &
-                                      -F6*fh(i+2,j,k)+    fh(i+3,j,k))
-  ! set imax and imin_lopsided 0
-      endif
-    endif
-
-  ! y direction   
-      if(Sfy(i,j,k) > ZEO)then
-        if(j+3 <= jmax)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfy(i,j,k)*d12dy*(-F3*fh(i,j-1,k)-F10*fh(i,j,k)+F18*fh(i,j+1,k) &
-                                      -F6*fh(i,j+2,k)+    fh(i,j+3,k))
-      elseif(j+2 <= jmax)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
-
-      elseif(j+1 <= jmax)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfy(i,j,k)*d12dy*(-F3*fh(i,j+1,k)-F10*fh(i,j,k)+F18*fh(i,j-1,k) &
-                                      -F6*fh(i,j-2,k)+    fh(i,j-3,k))
-  ! set imax and imin_lopsided 0
-      endif
-    elseif(Sfy(i,j,k) < ZEO)then
-        if(j-3 >= jmin_lopsided)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfy(i,j,k)*d12dy*(-F3*fh(i,j+1,k)-F10*fh(i,j,k)+F18*fh(i,j-1,k) &
-                                      -F6*fh(i,j-2,k)+    fh(i,j-3,k))
-      elseif(j-2 >= jmin_lopsided)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
-
-      elseif(j-1 >= jmin_lopsided)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfy(i,j,k)*d12dy*(-F3*fh(i,j-1,k)-F10*fh(i,j,k)+F18*fh(i,j+1,k) &
-                                      -F6*fh(i,j+2,k)+    fh(i,j+3,k))
-  ! set jmax and jmin_lopsided 0
-      endif
-    endif
-
-  ! z direction   
-      if(Sfz(i,j,k) > ZEO)then
-        if(k+3 <= kmax)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k-1)-F10*fh(i,j,k)+F18*fh(i,j,k+1) &
-                                      -F6*fh(i,j,k+2)+    fh(i,j,k+3))
-      elseif(k+2 <= kmax)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
-
-      elseif(k+1 <= kmax)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k+1)-F10*fh(i,j,k)+F18*fh(i,j,k-1) &
-                                      -F6*fh(i,j,k-2)+    fh(i,j,k-3))
-  ! set imax and imin_lopsided 0
-      endif
-    elseif(Sfz(i,j,k) < ZEO)then
-        if(k-3 >= kmin_lopsided)then
-  !         v
-  ! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
-  !  i     12dx       i-v      i      i+v     i+2v    i+3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
-                    Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k+1)-F10*fh(i,j,k)+F18*fh(i,j,k-1) &
-                                      -F6*fh(i,j,k-2)+    fh(i,j,k-3))
-      elseif(k-2 >= kmin_lopsided)then
-  !
-  !              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
-  !  fx(i) = ---------------------------------------------
-  !                             12 dx
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
-                    Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
-
-      elseif(k-1 >= kmin_lopsided)then
-  !         v
-  ! D f = ------[   3f    + 10f  - 18f    + 6f     - f     ]
-  !  i     12dx       i+v      i      i-v     i-2v    i-3v
-      f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
-                    Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k-1)-F10*fh(i,j,k)+F18*fh(i,j,k+1) &
-                                      -F6*fh(i,j,k+2)+    fh(i,j,k+3))
-  ! set kmax and kmin_lopsided 0
-      endif
-    endif
-
-
-    if(i-3 >= imin_kodis .and. i+3 <= imax .and. &
-      j-3 >= jmin_kodis .and. j+3 <= jmax .and. &
-      k-3 >= kmin_kodis .and. k+3 <= kmax) then
-
-  ! calculation order if important ?
-    f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/cof *( (     &
-                                (fh(i-3,j,k)+fh(i+3,j,k)) - &
-                            SIX*(fh(i-2,j,k)+fh(i+2,j,k)) + &
-                            FIT*(fh(i-1,j,k)+fh(i+1,j,k)) - &
-                            TWT* fh(i,j,k)            )/dX + &
-                                                    (     &
-                                (fh(i,j-3,k)+fh(i,j+3,k)) - &
-                            SIX*(fh(i,j-2,k)+fh(i,j+2,k)) + &
-                            FIT*(fh(i,j-1,k)+fh(i,j+1,k)) - &
-                            TWT* fh(i,j,k)            )/dY + &
-                                                    (     &
-                                (fh(i,j,k-3)+fh(i,j,k+3)) - &
-                            SIX*(fh(i,j,k-2)+fh(i,j,k+2)) + &
-                            FIT*(fh(i,j,k-1)+fh(i,j,k+1)) - &
-                            TWT* fh(i,j,k)            )/dZ )
-
-    endif
-
-    enddo
-    enddo
-    enddo
-
-    return
-
-
-
-  end subroutine merge_lopsided_kodis
--- a/AMSS_NCKU_source/diff_new.f90
+++ b/AMSS_NCKU_source/diff_new.f90
@@ -1000,7 +1000,86 @@
  do k=1,ex(3)-1
  do j=1,ex(2)-1
  do i=1,ex(1)-1
+#if 0  
+! x direction   
+        if(i+2 <= imax .and. i-2 >= imin)then
+!
+!              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
+!  fx(i) = ---------------------------------------------
+!                             12 dx
+      fx(i,j,k)=d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))

+    elseif(i+1 <= imax .and. i-1 >= imin)then
+!
+!              - f(i-1) + f(i+1)
+!  fx(i) = --------------------------------
+!                     2 dx
+      fx(i,j,k)=d2dx*(-fh(i-1,j,k)+fh(i+1,j,k))
+
+! set imax and imin 0
+    endif
+! y direction   
+        if(j+2 <= jmax .and. j-2 >= jmin)then
+
+      fy(i,j,k)=d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+    elseif(j+1 <= jmax .and. j-1 >= jmin)then
+
+     fy(i,j,k)=d2dy*(-fh(i,j-1,k)+fh(i,j+1,k))
+
+! set jmax and jmin 0
+    endif
+! z direction   
+        if(k+2 <= kmax .and. k-2 >= kmin)then
+
+      fz(i,j,k)=d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+    elseif(k+1 <= kmax .and. k-1 >= kmin)then
+
+      fz(i,j,k)=d2dz*(-fh(i,j,k-1)+fh(i,j,k+1))
+
+! set kmax and kmin 0
+    endif
+#elif 0
+! x direction   
+        if(i+2 <= imax .and. i-2 >= imin)then
+!
+!              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
+!  fx(i) = ---------------------------------------------
+!                             12 dx
+      fx(i,j,k)=d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+    elseif(i+3 <= imax .and. i-1 >= imin)then
+      fx(i,j,k)=d12dx*(-3.d0*fh(i-1,j,k)-1.d1*fh(i,j,k)+1.8d1*fh(i+1,j,k)-6.d0*fh(i+2,j,k)+fh(i+3,j,k))
+    elseif(i+1 <= imax .and. i-3 >= imin)then
+      fx(i,j,k)=d12dx*( 3.d0*fh(i+1,j,k)+1.d1*fh(i,j,k)-1.8d1*fh(i-1,j,k)+6.d0*fh(i-2,j,k)-fh(i-3,j,k))
+! set imax and imin 0
+    endif
+! y direction   
+        if(j+2 <= jmax .and. j-2 >= jmin)then
+
+      fy(i,j,k)=d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+    elseif(j+3 <= jmax .and. j-1 >= jmin)then
+      fy(i,j,k)=d12dy*(-3.d0*fh(i,j-1,k)-1.d1*fh(i,j,k)+1.8d1*fh(i,j+1,k)-6.d0*fh(i,j+2,k)+fh(i,j+3,k))
+    elseif(j+1 <= jmax .and. j-3 >= jmin)then
+      fy(i,j,k)=d12dy*( 3.d0*fh(i,j+1,k)+1.d1*fh(i,j,k)-1.8d1*fh(i,j-1,k)+6.d0*fh(i,j-2,k)-fh(i,j-3,k))
+
+! set jmax and jmin 0
+    endif
+! z direction   
+        if(k+2 <= kmax .and. k-2 >= kmin)then
+
+      fz(i,j,k)=d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+    elseif(k+3 <= kmax .and. k-1 >= kmin)then
+      fz(i,j,k)=d12dz*(-3.d0*fh(i,j,k-1)-1.d1*fh(i,j,k)+1.8d1*fh(i,j,k+1)-6.d0*fh(i,j,k+2)+fh(i,j,k+3))
+    elseif(k+1 <= kmax .and. k-3 >= kmin)then
+      fz(i,j,k)=d12dz*( 3.d0*fh(i,j,k+1)+1.d1*fh(i,j,k)-1.8d1*fh(i,j,k-1)+6.d0*fh(i,j,k-2)-fh(i,j,k-3))
+
+! set kmax and kmin 0
+    endif
+#else
 ! for bam comparison
   if(i+2 <= imax .and. i-2 >= imin .and. &
      j+2 <= jmax .and. j-2 >= jmin .and. &
@@ -1015,7 +1094,7 @@
      fy(i,j,k)=d2dy*(-fh(i,j-1,k)+fh(i,j+1,k))
      fz(i,j,k)=d2dz*(-fh(i,j,k-1)+fh(i,j,k+1))
   endif
-
+#endif
  enddo
  enddo
  enddo
@@ -1325,7 +1404,85 @@
  do k=1,ex(3)-1
  do j=1,ex(2)-1
  do i=1,ex(1)-1
+#if 0  
+!~~~~~~ fxx
+        if(i+2 <= imax .and. i-2 >= imin)then
+!
+!               - f(i-2) + 16 f(i-1) - 30 f(i) + 16 f(i+1) - f(i+2)
+!  fxx(i) = ----------------------------------------------------------
+!                                  12 dx^2 
+   fxx(i,j,k) = Fdxdx*(-fh(i-2,j,k)+F16*fh(i-1,j,k)-F30*fh(i,j,k) &
+                       -fh(i+2,j,k)+F16*fh(i+1,j,k)              )
+   elseif(i+1 <= imax .and. i-1 >= imin)then
+!
+!               f(i-1) - 2 f(i) + f(i+1)
+!  fxx(i) = --------------------------------
+!                         dx^2 
+   fxx(i,j,k) = Sdxdx*(fh(i-1,j,k)-TWO*fh(i,j,k) &
+                      +fh(i+1,j,k)              )
+   endif

+
+!~~~~~~ fyy
+        if(j+2 <= jmax .and. j-2 >= jmin)then
+
+   fyy(i,j,k) = Fdydy*(-fh(i,j-2,k)+F16*fh(i,j-1,k)-F30*fh(i,j,k) &
+                       -fh(i,j+2,k)+F16*fh(i,j+1,k)              )
+   elseif(j+1 <= jmax .and. j-1 >= jmin)then
+
+   fyy(i,j,k) = Sdydy*(fh(i,j-1,k)-TWO*fh(i,j,k) &
+                      +fh(i,j+1,k)              )
+   endif
+
+!~~~~~~ fzz
+        if(k+2 <= kmax .and. k-2 >= kmin)then
+
+   fzz(i,j,k) = Fdzdz*(-fh(i,j,k-2)+F16*fh(i,j,k-1)-F30*fh(i,j,k) &
+                       -fh(i,j,k+2)+F16*fh(i,j,k+1)              )
+   elseif(k+1 <= kmax .and. k-1 >= kmin)then
+
+   fzz(i,j,k) = Sdzdz*(fh(i,j,k-1)-TWO*fh(i,j,k) &
+                      +fh(i,j,k+1)              )
+   endif
+!~~~~~~ fxy
+       if(i+2 <= imax .and. i-2 >= imin .and. j+2 <= jmax .and. j-2 >= jmin)then
+!
+!                 ( f(i-2,j-2) - 8 f(i-1,j-2) + 8 f(i+1,j-2) - f(i+2,j-2) )
+!             - 8 ( f(i-2,j-1) - 8 f(i-1,j-1) + 8 f(i+1,j-1) - f(i+2,j-1) )
+!             + 8 ( f(i-2,j+1) - 8 f(i-1,j+1) + 8 f(i+1,j+1) - f(i+2,j+1) )
+!             -   ( f(i-2,j+2) - 8 f(i-1,j+2) + 8 f(i+1,j+2) - f(i+2,j+2) )
+!  fxy(i,j) = ----------------------------------------------------------------
+!                                  144 dx dy
+   fxy(i,j,k) = Fdxdy*(     (fh(i-2,j-2,k)-F8*fh(i-1,j-2,k)+F8*fh(i+1,j-2,k)-fh(i+2,j-2,k))  &
+                       -F8 *(fh(i-2,j-1,k)-F8*fh(i-1,j-1,k)+F8*fh(i+1,j-1,k)-fh(i+2,j-1,k))  &
+                       +F8 *(fh(i-2,j+1,k)-F8*fh(i-1,j+1,k)+F8*fh(i+1,j+1,k)-fh(i+2,j+1,k))  &
+                       -    (fh(i-2,j+2,k)-F8*fh(i-1,j+2,k)+F8*fh(i+1,j+2,k)-fh(i+2,j+2,k)))
+
+   elseif(i+1 <= imax .and. i-1 >= imin .and. j+1 <= jmax .and. j-1 >= jmin)then
+!                 f(i-1,j-1) - f(i+1,j-1) - f(i-1,j+1) + f(i+1,j+1) 
+!  fxy(i,j) = -----------------------------------------------------------
+!                                      4 dx dy
+   fxy(i,j,k) = Sdxdy*(fh(i-1,j-1,k)-fh(i+1,j-1,k)-fh(i-1,j+1,k)+fh(i+1,j+1,k))
+   endif
+!~~~~~~ fxz
+       if(i+2 <= imax .and. i-2 >= imin .and. k+2 <= kmax .and. k-2 >= kmin)then
+   fxz(i,j,k) = Fdxdz*(     (fh(i-2,j,k-2)-F8*fh(i-1,j,k-2)+F8*fh(i+1,j,k-2)-fh(i+2,j,k-2))  &
+                       -F8 *(fh(i-2,j,k-1)-F8*fh(i-1,j,k-1)+F8*fh(i+1,j,k-1)-fh(i+2,j,k-1))  &
+                       +F8 *(fh(i-2,j,k+1)-F8*fh(i-1,j,k+1)+F8*fh(i+1,j,k+1)-fh(i+2,j,k+1))  &
+                       -    (fh(i-2,j,k+2)-F8*fh(i-1,j,k+2)+F8*fh(i+1,j,k+2)-fh(i+2,j,k+2)))
+   elseif(i+1 <= imax .and. i-1 >= imin .and. k+1 <= kmax .and. k-1 >= kmin)then
+   fxz(i,j,k) = Sdxdz*(fh(i-1,j,k-1)-fh(i+1,j,k-1)-fh(i-1,j,k+1)+fh(i+1,j,k+1))
+   endif
+!~~~~~~ fyz
+       if(j+2 <= jmax .and. j-2 >= jmin .and. k+2 <= kmax .and. k-2 >= kmin)then
+   fyz(i,j,k) = Fdydz*(     (fh(i,j-2,k-2)-F8*fh(i,j-1,k-2)+F8*fh(i,j+1,k-2)-fh(i,j+2,k-2))  &
+                       -F8 *(fh(i,j-2,k-1)-F8*fh(i,j-1,k-1)+F8*fh(i,j+1,k-1)-fh(i,j+2,k-1))  &
+                       +F8 *(fh(i,j-2,k+1)-F8*fh(i,j-1,k+1)+F8*fh(i,j+1,k+1)-fh(i,j+2,k+1))  &
+                       -    (fh(i,j-2,k+2)-F8*fh(i,j-1,k+2)+F8*fh(i,j+1,k+2)-fh(i,j+2,k+2)))
+   elseif(j+1 <= jmax .and. j-1 >= jmin .and. k+1 <= kmax .and. k-1 >= kmin)then
+   fyz(i,j,k) = Sdydz*(fh(i,j-1,k-1)-fh(i,j+1,k-1)-fh(i,j-1,k+1)+fh(i,j+1,k+1))
+   endif 
+#else
 ! for bam comparison
   if(i+2 <= imax .and. i-2 >= imin .and. &
      j+2 <= jmax .and. j-2 >= jmin .and. &
@@ -1361,7 +1518,7 @@
   fxz(i,j,k) = Sdxdz*(fh(i-1,j,k-1)-fh(i+1,j,k-1)-fh(i-1,j,k+1)+fh(i+1,j,k+1))
   fyz(i,j,k) = Sdydz*(fh(i,j-1,k-1)-fh(i,j+1,k-1)-fh(i,j-1,k+1)+fh(i,j+1,k+1))
   endif
-
+#endif
   enddo
   enddo
   enddo
--- a/AMSS_NCKU_source/fmisc.f90
+++ b/AMSS_NCKU_source/fmisc.f90
@@ -326,7 +326,6 @@ subroutine symmetry_bd(ord,extc,func,funcc,SoA)

  funcc(1:extc(1),1:extc(2),1:extc(3)) = func
   do i=0,ord-1
-      
      funcc(-i,1:extc(2),1:extc(3)) = funcc(i+2,1:extc(2),1:extc(3))*SoA(1)
   enddo
   do i=0,ord-1
--- a/AMSS_NCKU_source/kodiss.f90
+++ b/AMSS_NCKU_source/kodiss.f90
@@ -6,6 +6,101 @@
 ! Vertex or Cell is distinguished in routine symmetry_bd which locates in
 ! file "fmisc.f90"

+#if (ghost_width == 2)
+! second order code
+
+!------------------------------------------------------------------------------------------------------------------------------
+!usual type Kreiss-Oliger type numerical dissipation
+!We support cell center only
+!  (D_+D_-)^2 =
+!   f(i-2) - 4 f(i-1) + 6 f(i) - 4 f(i+1) + f(i+2)
+! ------------------------------------------------------
+!                       dx^4
+!------------------------------------------------------------------------------------------------------------------------------
+! do not add dissipation near boundary
+subroutine kodis(ex,X,Y,Z,f,f_rhs,SoA,Symmetry,eps)
+
+implicit none
+! argument variables
+integer,intent(in) :: Symmetry
+integer,dimension(3),intent(in)::ex
+real*8, dimension(1:3), intent(in) :: SoA
+double precision,intent(in),dimension(ex(1))::X
+double precision,intent(in),dimension(ex(2))::Y
+double precision,intent(in),dimension(ex(3))::Z
+double precision,intent(in),dimension(ex(1),ex(2),ex(3))::f
+double precision,intent(inout),dimension(ex(1),ex(2),ex(3))::f_rhs
+real*8,intent(in) :: eps
+
+!~~~~~~ other variables
+
+  real*8 :: dX,dY,dZ
+  real*8,dimension(-1:ex(1),-1:ex(2),-1:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+  real*8,parameter   :: cof = 1.6d1 ! 2^4
+  real*8,  parameter :: F4=4.d0,F6=6.d0
+  integer::i,j,k
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -1
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -1
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -1
+
+  call symmetry_bd(2,ex,f,fh,SoA)
+
+!   f(i-2) - 4 f(i-1) + 6 f(i) - 4 f(i+1) + f(i+2)
+! ------------------------------------------------------
+!                       dx^4
+
+!  note the sign (-1)^r-1, now r=2
+  do k=1,ex(3)
+  do j=1,ex(2)
+  do i=1,ex(1)
+
+  if(i-2 >= imin .and. i+2 <= imax .and. &
+     j-2 >= jmin .and. j+2 <= jmax .and. &
+     k-2 >= kmin .and. k+2 <= kmax) then
+! x direction
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dX/cof * (     &
+                                (fh(i-2,j,k)+fh(i+2,j,k)) &
+                         - F4 * (fh(i-1,j,k)+fh(i+1,j,k)) &
+                         + F6 *  fh(i,j,k) )
+! y direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dY/cof * (     &
+                                (fh(i,j-2,k)+fh(i,j+2,k)) &
+                         - F4 * (fh(i,j-1,k)+fh(i,j+1,k)) &
+                         + F6 *  fh(i,j,k) )
+! z direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dZ/cof * (     &
+                                (fh(i,j,k-2)+fh(i,j,k+2)) &
+                         - F4 * (fh(i,j,k-1)+fh(i,j,k+1)) &
+                         + F6 *  fh(i,j,k) )
+
+  endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+end subroutine kodis
+
+#elif (ghost_width == 3)
 ! fourth order code

 !---------------------------------------------------------------------------------------------
@@ -61,7 +156,7 @@ integer, parameter :: NO_SYMM=0, OCTANT=2
  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -2
  if(Symmetry == OCTANT .and. dabs(X(1)) < dX) imin = -2
  if(Symmetry == OCTANT .and. dabs(Y(1)) < dY) jmin = -2
-  !print*,'imin,jmin,kmin=',imin,jmin,kmin
+
  call symmetry_bd(3,ex,f,fh,SoA)

  do k=1,ex(3)
@@ -71,7 +166,28 @@ integer, parameter :: NO_SYMM=0, OCTANT=2
  if(i-3 >= imin .and. i+3 <= imax .and. &
     j-3 >= jmin .and. j+3 <= jmax .and. &
     k-3 >= kmin .and. k+3 <= kmax) then
+#if 0     
+! x direction
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dX/cof * (     &
+                              (fh(i-3,j,k)+fh(i+3,j,k)) - &
+                          SIX*(fh(i-2,j,k)+fh(i+2,j,k)) + &
+                          FIT*(fh(i-1,j,k)+fh(i+1,j,k)) - &
+                          TWT* fh(i,j,k)            )
+! y direction

+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dY/cof * (     &
+                              (fh(i,j-3,k)+fh(i,j+3,k)) - &
+                          SIX*(fh(i,j-2,k)+fh(i,j+2,k)) + &
+                          FIT*(fh(i,j-1,k)+fh(i,j+1,k)) - &
+                          TWT* fh(i,j,k)            )
+! z direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dZ/cof * (     &
+                              (fh(i,j,k-3)+fh(i,j,k+3)) - &
+                          SIX*(fh(i,j,k-2)+fh(i,j,k+2)) + &
+                          FIT*(fh(i,j,k-1)+fh(i,j,k+1)) - &
+                          TWT* fh(i,j,k)            )
+#else
 ! calculation order if important ?
   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/cof *( (     &
                              (fh(i-3,j,k)+fh(i+3,j,k)) - &
@@ -88,6 +204,105 @@ integer, parameter :: NO_SYMM=0, OCTANT=2
                          SIX*(fh(i,j,k-2)+fh(i,j,k+2)) + &
                          FIT*(fh(i,j,k-1)+fh(i,j,k+1)) - &
                          TWT* fh(i,j,k)            )/dZ )
+#endif
+  endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+  end subroutine kodis
+
+#elif (ghost_width == 4)
+! sixth order code
+!------------------------------------------------------------------------------------------------------------------------------
+!usual type Kreiss-Oliger type numerical dissipation
+!We support cell center only
+!  (D_+D_-)^4 =
+!   f(i-4) - 8 f(i-3) + 28 f(i-2) - 56 f(i-1) + 70 f(i) - 56 f(i+1) + 28 f(i+2) - 8 f(i+3) + f(i+4)
+! ----------------------------------------------------------------------------------------------------------
+!                                              dx^8
+!------------------------------------------------------------------------------------------------------------------------------
+! do not add dissipation near boundary
+subroutine kodis(ex,X,Y,Z,f,f_rhs,SoA,Symmetry,eps)
+
+implicit none
+! argument variables
+integer,intent(in) :: Symmetry
+integer,dimension(3),intent(in)::ex
+real*8, dimension(1:3), intent(in) :: SoA
+double precision,intent(in),dimension(ex(1))::X
+double precision,intent(in),dimension(ex(2))::Y
+double precision,intent(in),dimension(ex(3))::Z
+double precision,intent(in),dimension(ex(1),ex(2),ex(3))::f
+double precision,intent(inout),dimension(ex(1),ex(2),ex(3))::f_rhs
+real*8,intent(in) :: eps
+
+!~~~~~~ other variables
+
+  real*8 :: dX,dY,dZ
+  real*8,dimension(-3:ex(1),-3:ex(2),-3:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+  real*8,parameter   :: cof = 2.56d2 ! 2^8
+  real*8,  parameter :: F8=8.d0,F28=2.8d1,F56=5.6d1,F70=7.d1
+  integer::i,j,k
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+  
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -3
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -3
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -3
+
+  call symmetry_bd(4,ex,f,fh,SoA)
+
+!   f(i-4) - 8 f(i-3) + 28 f(i-2) - 56 f(i-1) + 70 f(i) - 56 f(i+1) + 28 f(i+2) - 8 f(i+3) + f(i+4)
+! ----------------------------------------------------------------------------------------------------------
+!                                              dx^8
+
+!  note the sign (-1)^r-1, now r=4
+  do k=1,ex(3)
+  do j=1,ex(2)
+  do i=1,ex(1)
+
+  if(i>imin+3 .and. i < imax-3 .and. &
+     j>jmin+3 .and. j < jmax-3 .and. &
+     k>kmin+3 .and. k < kmax-3) then
+! x direction
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dX/cof * (     &
+                                (fh(i-4,j,k)+fh(i+4,j,k)) &
+                         - F8 * (fh(i-3,j,k)+fh(i+3,j,k)) &
+                         +F28 * (fh(i-2,j,k)+fh(i+2,j,k)) &
+                         -F56 * (fh(i-1,j,k)+fh(i+1,j,k)) &
+                         +F70 *  fh(i,j,k) )
+! y direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dY/cof * (     &
+                                (fh(i,j-4,k)+fh(i,j+4,k)) &
+                         - F8 * (fh(i,j-3,k)+fh(i,j+3,k)) &
+                         +F28 * (fh(i,j-2,k)+fh(i,j+2,k)) &
+                         -F56 * (fh(i,j-1,k)+fh(i,j+1,k)) &
+                         +F70 *  fh(i,j,k) )
+! z direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) - eps/dZ/cof * (     &
+                                (fh(i,j,k-4)+fh(i,j,k+4)) &
+                         - F8 * (fh(i,j,k-3)+fh(i,j,k+3)) &
+                         +F28 * (fh(i,j,k-2)+fh(i,j,k+2)) &
+                         -F56 * (fh(i,j,k-1)+fh(i,j,k+1)) &
+                         +F70 *  fh(i,j,k) )

  endif

@@ -99,6 +314,119 @@ integer, parameter :: NO_SYMM=0, OCTANT=2

 end subroutine kodis

+#elif (ghost_width == 5)
+! eighth order code
+!------------------------------------------------------------------------------------------------------------------------------
+!usual type Kreiss-Oliger type numerical dissipation
+!We support cell center only
+! Note the notation D_+ and D_- [P240 of B. Gustafsson, H.-O. Kreiss, and J. Oliger, Time
+! Dependent Problems and Difference Methods (Wiley, New York, 1995).]
+! D_+ = (f(i+1) - f(i))/h
+! D_- = (f(i) - f(i-1))/h
+! then we have D_+D_- = D_-D_+ = (f(i+1) - 2f(i) + f(i-1))/h^2
+! for nth order accurate finite difference code, we need r =n/2+1
+!              D_+^rD_-^r = (D_+D_-)^r 
+! following the tradiation of PRD 77, 024027 (BB's calibration paper, Eq.(64),
+!  correct some typo according to above book) :
+! + eps*(-1)^(r-1)*h^(2r-1)/2^(2r)*(D_+D_-)^r 
+!
+!
+! this is for 8th order accurate finite difference scheme
+!  (D_+D_-)^5 =
+!  f(i-5) - 10 f(i-4) + 45 f(i-3) - 120 f(i-2) + 210 f(i-1) - 252 f(i) + 210 f(i+1) - 120 f(i+2) + 45 f(i+3) - 10 f(i+4) + f(i+5)
+! -------------------------------------------------------------------------------------------------------------------------------
+!                                                              dx^10
+!---------------------------------------------------------------------------------------------------------------------------------
+! do not add dissipation near boundary
+subroutine kodis(ex,X,Y,Z,f,f_rhs,SoA,Symmetry,eps)

+implicit none
+! argument variables
+integer,intent(in) :: Symmetry
+integer,dimension(3),intent(in)::ex
+real*8, dimension(1:3), intent(in) :: SoA
+double precision,intent(in),dimension(ex(1))::X
+double precision,intent(in),dimension(ex(2))::Y
+double precision,intent(in),dimension(ex(3))::Z
+double precision,intent(in),dimension(ex(1),ex(2),ex(3))::f
+double precision,intent(inout),dimension(ex(1),ex(2),ex(3))::f_rhs
+real*8,intent(in) :: eps

+!~~~~~~ other variables

+  real*8 :: dX,dY,dZ
+  real*8,dimension(-4:ex(1),-4:ex(2),-4:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+  real*8,parameter   :: cof = 1.024d3 ! 2^2r = 2^10
+  real*8,  parameter :: F10=1.d1,F45=4.5d1,F120=1.2d2,F210=2.1d2,F252=2.52d2
+  integer::i,j,k
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+  
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -4
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -4
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -4
+
+  call symmetry_bd(5,ex,f,fh,SoA)
+
+!  f(i-5) - 10 f(i-4) + 45 f(i-3) - 120 f(i-2) + 210 f(i-1) - 252 f(i) + 210 f(i+1) - 120 f(i+2) + 45 f(i+3) - 10 f(i+4) + f(i+5)
+! -------------------------------------------------------------------------------------------------------------------------------
+!                                                              dx^10
+
+!  note the sign (-1)^r-1, now r=5
+  do k=1,ex(3)
+  do j=1,ex(2)
+  do i=1,ex(1)
+
+  if(i>imin+4 .and. i < imax-4 .and. &
+     j>jmin+4 .and. j < jmax-4 .and. &
+     k>kmin+4 .and. k < kmax-4) then
+! x direction
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dX/cof * (      &
+                                 (fh(i-5,j,k)+fh(i+5,j,k)) &
+                         - F10 * (fh(i-4,j,k)+fh(i+4,j,k)) &
+                         + F45 * (fh(i-3,j,k)+fh(i+3,j,k)) &
+                         - F120* (fh(i-2,j,k)+fh(i+2,j,k)) &
+                         + F210* (fh(i-1,j,k)+fh(i+1,j,k)) &
+                         - F252 * fh(i,j,k) )
+! y direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dY/cof * (      &
+                                 (fh(i,j-5,k)+fh(i,j+5,k)) &
+                         - F10 * (fh(i,j-4,k)+fh(i,j+4,k)) &
+                         + F45 * (fh(i,j-3,k)+fh(i,j+3,k)) &
+                         - F120* (fh(i,j-2,k)+fh(i,j+2,k)) &
+                         + F210* (fh(i,j-1,k)+fh(i,j+1,k)) &
+                         - F252 * fh(i,j,k) )
+! z direction
+
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/dZ/cof * (      &
+                                 (fh(i,j,k-5)+fh(i,j,k+5)) &
+                         - F10 * (fh(i,j,k-4)+fh(i,j,k+4)) &
+                         + F45 * (fh(i,j,k-3)+fh(i,j,k+3)) &
+                         - F120* (fh(i,j,k-2)+fh(i,j,k+2)) &
+                         + F210* (fh(i,j,k-1)+fh(i,j,k+1)) &
+                         - F252 * fh(i,j,k) )
+
+  endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+end subroutine kodis
+
+#endif  
--- a/AMSS_NCKU_source/lopsidediff.f90
+++ b/AMSS_NCKU_source/lopsidediff.f90
@@ -7,7 +7,163 @@
 ! Vertex or Cell is distinguished in routine symmetry_bd which locates in
 ! file "fmisc.f90"

+#if (ghost_width == 2)
+! second order code

+!-----------------------------------------------------------------------------
+!         v
+! D f = ------[ - 3 f  + 4 f   - f     ]
+!  i     2dx         i      i+v   i+2v
+!
+! where
+!
+!        i
+!      |B |
+! v = -----
+!        i
+!       B
+!
+!-----------------------------------------------------------------------------
+subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
+  implicit none
+
+!~~~~~~> Input parameters:
+
+  integer, intent(in)  :: ex(1:3),Symmetry
+  real*8,  intent(in)  :: X(1:ex(1)),Y(1:ex(2)),Z(1:ex(3))
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(in)   :: f,Sfx,Sfy,Sfz
+
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(inout):: f_rhs
+  real*8,dimension(3),intent(in) ::SoA
+
+!~~~~~~> local variables:
+! note index -1,0, so we have 2 extra points
+  real*8,dimension(-1:ex(1),-1:ex(2),-1:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax,i,j,k
+  real*8 :: dX,dY,dZ
+  real*8 :: d2dx,d2dy,d2dz
+  real*8,  parameter :: ZEO=0.d0,ONE=1.d0,TWO=2.d0,THR=3.d0,FOUR=4.d0
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+
+  d2dx = ONE/TWO/dX
+  d2dy = ONE/TWO/dY
+  d2dz = ONE/TWO/dZ
+
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -1
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -1
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -1
+
+  call symmetry_bd(2,ex,f,fh,SoA)
+
+! upper bound set ex-1 only for efficiency, 
+! the loop body will set ex 0 also
+  do k=1,ex(3)-1
+  do j=1,ex(2)-1
+  do i=1,ex(1)-1
+! x direction   
+    if(Sfx(i,j,k) >= ZEO)then
+       if( i+2 <= imax .and. i >= imin)then
+!         v
+! D f = ------[ - 3 f  + 4 f   - f     ]
+!  i     2dx         i      i+v   i+2v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfx(i,j,k)*d2dx*(-THR*fh(i,j,k)+FOUR*fh(i+1,j,k)-fh(i+2,j,k))
+       elseif(i+1 <= imax .and. i >= imin)then
+!         v
+! D f = ------[ - f  + f   ]
+!  i      dx       i    i+v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfx(i,j,k)*d2dx*(-fh(i,j,k)+fh(i+1,j,k))
+
+       endif
+
+    elseif(Sfx(i,j,k) <= ZEO)then
+      if( i-2 >= imin .and. i <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfx(i,j,k)*d2dx*(-THR*fh(i,j,k)+FOUR*fh(i-1,j,k)-fh(i-2,j,k))
+      elseif(i-1 >= imin .and. i <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfx(i,j,k)*d2dx*(-fh(i,j,k)+fh(i-1,j,k))
+      endif
+
+! set imax and imin 0
+    endif
+
+! y direction   
+    if(Sfy(i,j,k) >= ZEO)then
+       if( j+2 <= jmax .and. j >= jmin)then
+!         v
+! D f = ------[ - 3 f  + 4 f   - f     ]
+!  i     2dx         i      i+v   i+2v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfy(i,j,k)*d2dy*(-THR*fh(i,j,k)+FOUR*fh(i,j+1,k)-fh(i,j+2,k))
+       elseif(j+1 <= jmax .and. j >= jmin)then
+!         v
+! D f = ------[ - f  + f   ]
+!  i      dx       i    i+v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfy(i,j,k)*d2dy*(-fh(i,j,k)+fh(i,j+1,k))
+       endif
+
+    elseif(Sfy(i,j,k) <= ZEO)then
+      if( j-2 >= jmin .and. j <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfy(i,j,k)*d2dy*(-THR*fh(i,j,k)+FOUR*fh(i,j-1,k)-fh(i,j-2,k))
+      elseif(j-1 >= jmin .and. j <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfy(i,j,k)*d2dy*(-fh(i,j,k)+fh(i,j-1,k))
+      endif
+
+! set jmin and jmax 0
+     endif
+!! z direction   
+    if(Sfz(i,j,k) >= ZEO)then
+      if( k+2 <= kmax .and. k >= kmin)then
+!         v
+! D f = ------[ - 3 f  + 4 f   - f     ]
+!  i     2dx         i      i+v   i+2v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfz(i,j,k)*d2dz*(-THR*fh(i,j,k)+FOUR*fh(i,j,k+1)-fh(i,j,k+2))
+       elseif(k+1 <= kmax .and. k >= kmin)then
+!         v
+! D f = ------[ - f  + f   ]
+!  i      dx       i    i+v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                           &
+                  Sfz(i,j,k)*d2dz*(-fh(i,j,k)+fh(i,j,k+1))
+       endif
+
+    elseif(Sfz(i,j,k) <= ZEO)then
+      if( k-2 >= kmin .and. k <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfz(i,j,k)*d2dz*(-THR*fh(i,j,k)+FOUR*fh(i,j,k-1)-fh(i,j,k-2))
+      elseif(k-1 >= kmin .and. k <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                           &
+                  Sfz(i,j,k)*d2dz*(-fh(i,j,k)+fh(i,j,k-1))
+      endif
+
+! set kmin and kmax 0
+     endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+  end subroutine lopsided
+
+#elif (ghost_width == 3)
 ! fourth order code

 !-----------------------------------------------------------------------------
@@ -80,7 +236,89 @@ subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
  do k=1,ex(3)-1
  do j=1,ex(2)-1
  do i=1,ex(1)-1
+#if 0  
+!! old code
+! x direction   
+    if(Sfx(i,j,k) >= ZEO .and. i+3 <= imax .and. i-1 >= imin)then
+!         v
+! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
+!  i     12dx       i-v      i      i+v     i+2v    i+3v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i-1,j,k)-F10*fh(i,j,k)+F18*fh(i+1,j,k) &
+                                    -F6*fh(i+2,j,k)+    fh(i+3,j,k))

+    elseif(Sfx(i,j,k) <= ZEO .and. i-3 >= imin .and. i+1 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i+1,j,k)-F10*fh(i,j,k)+F18*fh(i-1,j,k) &
+                                    -F6*fh(i-2,j,k)+    fh(i-3,j,k))
+
+     elseif(i+2 <= imax .and. i-2 >= imin)then
+!
+!              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
+!  fx(i) = ---------------------------------------------
+!                             12 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+     elseif(i+1 <= imax .and. i-1 >= imin)then
+!
+!              - f(i-1) + f(i+1)
+!  fx(i) = --------------------------------
+!                     2 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfx(i,j,k)*d2dx*(-fh(i-1,j,k)+fh(i+1,j,k))
+
+! set imax and imin 0
+    endif
+
+! y direction   
+    if(Sfy(i,j,k) >= ZEO .and. j+3 <= jmax .and. j-1 >= jmin)then
+!         v
+! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
+!  i     12dx       i-v      i      i+v     i+2v    i+3v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j-1,k)-F10*fh(i,j,k)+F18*fh(i,j+1,k) &
+                                    -F6*fh(i,j+2,k)+    fh(i,j+3,k))
+
+    elseif(Sfy(i,j,k) <= ZEO .and. j-3 >= jmin .and. j+1 <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j+1,k)-F10*fh(i,j,k)+F18*fh(i,j-1,k) &
+                                    -F6*fh(i,j-2,k)+    fh(i,j-3,k))
+
+     elseif(j+2 <= jmax .and. j-2 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+     elseif(j+1 <= jmax .and. j-1 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfy(i,j,k)*d2dy*(-fh(i,j-1,k)+fh(i,j+1,k))
+! set jmin and jmax 0
+     endif
+!! z direction   
+    if(Sfz(i,j,k) >= ZEO .and. k+3 <= kmax .and. k-1 >= kmin)then
+!         v
+! D f = ------[ - 3f    - 10f  + 18f    - 6f     + f     ]
+!  i     12dx       i-v      i      i+v     i+2v    i+3v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k-1)-F10*fh(i,j,k)+F18*fh(i,j,k+1) &
+                                    -F6*fh(i,j,k+2)+    fh(i,j,k+3))
+
+    elseif(Sfz(i,j,k) <= ZEO .and. k-3 >= kmin .and. k+1 <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k+1)-F10*fh(i,j,k)+F18*fh(i,j,k-1) &
+                                    -F6*fh(i,j,k-2)+    fh(i,j,k-3))
+
+     elseif(k+2 <= kmax .and. k-2 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+     elseif(k+1 <= kmax .and. k-1 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+Sfz(i,j,k)*d2dz*(-fh(i,j,k-1)+fh(i,j,k+1))
+! set kmin and kmax 0
+     endif
+#else
 !! new code, 2012dec27, based on bam
 ! x direction   
    if(Sfx(i,j,k) > ZEO)then
@@ -240,6 +478,7 @@ subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
 ! set kmax and kmin 0
     endif
   endif
+#endif
  enddo
  enddo
  enddo
@@ -247,3 +486,612 @@ subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
  return

  end subroutine lopsided
+
+!-----------------------------------------------------------------------------
+! Combined advection (lopsided) + Kreiss-Oliger dissipation (kodis)
+! Shares the symmetry_bd buffer fh, eliminating one full-grid copy per call.
+! Mathematically identical to calling lopsided then kodis separately.
+!-----------------------------------------------------------------------------
+subroutine lopsided_kodis(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA,eps)
+  implicit none
+
+!~~~~~~> Input parameters:
+
+  integer, intent(in)  :: ex(1:3),Symmetry
+  real*8,  intent(in)  :: X(1:ex(1)),Y(1:ex(2)),Z(1:ex(3))
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(in)   :: f,Sfx,Sfy,Sfz
+
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(inout):: f_rhs
+  real*8,dimension(3),intent(in) ::SoA
+  real*8,intent(in) :: eps
+
+!~~~~~~> local variables:
+! note index -2,-1,0, so we have 3 extra points
+  real*8,dimension(-2:ex(1),-2:ex(2),-2:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax,i,j,k
+  real*8 :: dX,dY,dZ
+  real*8 :: d12dx,d12dy,d12dz,d2dx,d2dy,d2dz
+  real*8,  parameter :: ZEO=0.d0,ONE=1.d0, F3=3.d0
+  real*8,  parameter :: TWO=2.d0,F6=6.0d0,F18=1.8d1
+  real*8,  parameter :: F12=1.2d1, F10=1.d1,EIT=8.d0
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+! kodis parameters
+  real*8, parameter :: SIX=6.d0,FIT=1.5d1,TWT=2.d1
+  real*8, parameter :: cof=6.4d1   ! 2^6
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+
+  d12dx = ONE/F12/dX
+  d12dy = ONE/F12/dY
+  d12dz = ONE/F12/dZ
+
+  d2dx = ONE/TWO/dX
+  d2dy = ONE/TWO/dY
+  d2dz = ONE/TWO/dZ
+
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -2
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -2
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -2
+
+! Single symmetry_bd call shared by both advection and dissipation
+  call symmetry_bd(3,ex,f,fh,SoA)
+
+! ---- Advection (lopsided) loop ----
+! upper bound set ex-1 only for efficiency, 
+! the loop body will set ex 0 also
+  do k=1,ex(3)-1
+  do j=1,ex(2)-1
+  do i=1,ex(1)-1
+! x direction   
+    if(Sfx(i,j,k) > ZEO)then
+      if(i+3 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i-1,j,k)-F10*fh(i,j,k)+F18*fh(i+1,j,k) &
+                                    -F6*fh(i+2,j,k)+    fh(i+3,j,k))
+     elseif(i+2 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+     elseif(i+1 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i+1,j,k)-F10*fh(i,j,k)+F18*fh(i-1,j,k) &
+                                    -F6*fh(i-2,j,k)+    fh(i-3,j,k))
+     endif
+   elseif(Sfx(i,j,k) < ZEO)then
+      if(i-3 >= imin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i+1,j,k)-F10*fh(i,j,k)+F18*fh(i-1,j,k) &
+                                    -F6*fh(i-2,j,k)+    fh(i-3,j,k))
+     elseif(i-2 >= imin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+     elseif(i-1 >= imin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfx(i,j,k)*d12dx*(-F3*fh(i-1,j,k)-F10*fh(i,j,k)+F18*fh(i+1,j,k) &
+                                    -F6*fh(i+2,j,k)+    fh(i+3,j,k))
+     endif
+   endif
+
+! y direction   
+    if(Sfy(i,j,k) > ZEO)then
+      if(j+3 <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j-1,k)-F10*fh(i,j,k)+F18*fh(i,j+1,k) &
+                                    -F6*fh(i,j+2,k)+    fh(i,j+3,k))
+     elseif(j+2 <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+     elseif(j+1 <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j+1,k)-F10*fh(i,j,k)+F18*fh(i,j-1,k) &
+                                    -F6*fh(i,j-2,k)+    fh(i,j-3,k))
+     endif
+   elseif(Sfy(i,j,k) < ZEO)then
+      if(j-3 >= jmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j+1,k)-F10*fh(i,j,k)+F18*fh(i,j-1,k) &
+                                    -F6*fh(i,j-2,k)+    fh(i,j-3,k))
+     elseif(j-2 >= jmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+     elseif(j-1 >= jmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfy(i,j,k)*d12dy*(-F3*fh(i,j-1,k)-F10*fh(i,j,k)+F18*fh(i,j+1,k) &
+                                    -F6*fh(i,j+2,k)+    fh(i,j+3,k))
+     endif
+   endif
+
+! z direction   
+    if(Sfz(i,j,k) > ZEO)then
+      if(k+3 <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k-1)-F10*fh(i,j,k)+F18*fh(i,j,k+1) &
+                                    -F6*fh(i,j,k+2)+    fh(i,j,k+3))
+     elseif(k+2 <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+     elseif(k+1 <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k+1)-F10*fh(i,j,k)+F18*fh(i,j,k-1) &
+                                    -F6*fh(i,j,k-2)+    fh(i,j,k-3))
+     endif
+   elseif(Sfz(i,j,k) < ZEO)then
+      if(k-3 >= kmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k+1)-F10*fh(i,j,k)+F18*fh(i,j,k-1) &
+                                    -F6*fh(i,j,k-2)+    fh(i,j,k-3))
+     elseif(k-2 >= kmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+     elseif(k-1 >= kmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                   &
+                  Sfz(i,j,k)*d12dz*(-F3*fh(i,j,k-1)-F10*fh(i,j,k)+F18*fh(i,j,k+1) &
+                                    -F6*fh(i,j,k+2)+    fh(i,j,k+3))
+     endif
+   endif
+  enddo
+  enddo
+  enddo
+
+! ---- Dissipation (kodis) loop ----
+  if(eps > ZEO) then
+  do k=1,ex(3)
+  do j=1,ex(2)
+  do i=1,ex(1)
+
+  if(i-3 >= imin .and. i+3 <= imax .and. &
+     j-3 >= jmin .and. j+3 <= jmax .and. &
+     k-3 >= kmin .and. k+3 <= kmax) then
+   f_rhs(i,j,k)       = f_rhs(i,j,k) + eps/cof *( (     &
+                              (fh(i-3,j,k)+fh(i+3,j,k)) - &
+                          SIX*(fh(i-2,j,k)+fh(i+2,j,k)) + &
+                          FIT*(fh(i-1,j,k)+fh(i+1,j,k)) - &
+                          TWT* fh(i,j,k)            )/dX + &
+                                                  (     &
+                              (fh(i,j-3,k)+fh(i,j+3,k)) - &
+                          SIX*(fh(i,j-2,k)+fh(i,j+2,k)) + &
+                          FIT*(fh(i,j-1,k)+fh(i,j+1,k)) - &
+                          TWT* fh(i,j,k)            )/dY + &
+                                                  (     &
+                              (fh(i,j,k-3)+fh(i,j,k+3)) - &
+                          SIX*(fh(i,j,k-2)+fh(i,j,k+2)) + &
+                          FIT*(fh(i,j,k-1)+fh(i,j,k+1)) - &
+                          TWT* fh(i,j,k)            )/dZ )
+  endif
+
+  enddo
+  enddo
+  enddo
+  endif
+
+  return
+
+  end subroutine lopsided_kodis
+
+#elif (ghost_width == 4)
+! sixth order code
+! Compute advection terms in right hand sides of field equations
+!         v
+! D f = ------[ 2f     - 24f    - 35f  + 80f    - 30f     + 8f     - f    ]
+!  i     60dx     i-2v      i-v      i      i+v      i+2v     i+3v    i+4v
+!
+! where
+!
+!        i
+!      |B |
+! v = -----
+!        i
+!       B
+!
+!-----------------------------------------------------------------------------
+subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
+  implicit none
+
+!~~~~~~> Input parameters:
+
+  integer, intent(in)  :: ex(1:3),Symmetry
+  real*8,  intent(in)  :: X(1:ex(1)),Y(1:ex(2)),Z(1:ex(3))
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(in)   :: f,Sfx,Sfy,Sfz
+
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(inout):: f_rhs
+  real*8,dimension(3),intent(in) ::SoA
+
+!~~~~~~> local variables:
+
+  real*8,dimension(-3:ex(1),-3:ex(2),-3:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax,i,j,k
+  real*8 :: dX,dY,dZ
+  real*8 :: d60dx,d60dy,d60dz,d12dx,d12dy,d12dz,d2dx,d2dy,d2dz
+  real*8,  parameter :: ZEO=0.d0,ONE=1.d0, F60=6.d1
+  real*8,  parameter :: TWO=2.d0,F24=2.4d1,F35=3.5d1,F80=8.d1,F30=3.d1,EIT=8.d0
+  real*8,  parameter ::  F9=9.d0,F45=4.5d1,F12=1.2d1
+  real*8,  parameter ::  F10=1.d1,F77=7.7d1,F150=1.5d2,F100=1.d2,F50=5.d1,F15=1.5d1
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+
+  d60dx = ONE/F60/dX
+  d60dy = ONE/F60/dY
+  d60dz = ONE/F60/dZ
+
+  d12dx = ONE/F12/dX
+  d12dy = ONE/F12/dY
+  d12dz = ONE/F12/dZ
+
+  d2dx = ONE/TWO/dX
+  d2dy = ONE/TWO/dY
+  d2dz = ONE/TWO/dZ
+
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -3
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -3
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -3
+
+  call symmetry_bd(4,ex,f,fh,SoA)
+
+! upper bound set ex-1 only for efficiency, 
+! the loop body will set ex 0 also
+  do k=1,ex(3)-1
+  do j=1,ex(2)-1
+  do i=1,ex(1)-1
+! x direction   
+    if(Sfx(i,j,k) >= ZEO .and. i+4 <= imax .and. i-2 >= imin)then
+!         v
+! D f = ------[ 2f     - 24f    - 35f  + 80f    - 30f     + 8f     - f    ]
+!  i     60dx     i-2v      i-v      i      i+v      i+2v     i+3v    i+4v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                             &
+                  Sfx(i,j,k)*d60dx*(TWO*fh(i-2,j,k)-F24*fh(i-1,j,k)-F35*fh(i,j,k)+F80*fh(i+1,j,k) &
+                                   -F30*fh(i+2,j,k)+EIT*fh(i+3,j,k)-    fh(i+4,j,k))
+    elseif(Sfx(i,j,k) >= ZEO .and. i+5 <= imax .and. i-1 >= imin)then
+!         v
+! D f = ------[-10f    - 77f  + 150f    - 100f     + 50f     -15f     + 2f    ]
+!  i     60dx      i-v      i       i+v       i+2v      i+3v     i+4v    i+5v
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                        &
+                  Sfx(i,j,k)*d60dx*(-F10*fh(i-1,j,k)-F77*fh(i  ,j,k)+F150*fh(i+1,j,k)-F100*fh(i+2,j,k) &
+                                    +F50*fh(i+3,j,k)-F15*fh(i+4,j,k)+ TWO*fh(i+5,j,k))
+
+    elseif(Sfx(i,j,k) <= ZEO .and. i-4 >= imin .and. i+2 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                   &
+                  Sfx(i,j,k)*d60dx*(TWO*fh(i+2,j,k)-F24*fh(i+1,j,k)-F35*fh(i,j,k)+F80*fh(i-1,j,k) &
+                                   -F30*fh(i-2,j,k)+EIT*fh(i-3,j,k)-    fh(i-4,j,k))
+    elseif(Sfx(i,j,k) <= ZEO .and. i-5 >= imin .and. i+1 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                        &
+                  Sfx(i,j,k)*d60dx*(-F10*fh(i+1,j,k)-F77*fh(i  ,j,k)+F150*fh(i-1,j,k)-F100*fh(i-2,j,k) &
+                                    +F50*fh(i-3,j,k)-F15*fh(i-4,j,k)+ TWO*fh(i-5,j,k))
+
+     elseif(i+3 <= imax .and. i-3 >= imin)then
+!           - f(i-3) + 9 f(i-2) - 45 f(i-1) + 45 f(i+1) - 9 f(i+2) + f(i+3)
+!  fx(i) = -----------------------------------------------------------------
+!                                        60 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                              &
+                  Sfx(i,j,k)*d60dx*(-fh(i-3,j,k)+F9*fh(i-2,j,k)-F45*fh(i-1,j,k)+F45*fh(i+1,j,k)-F9*fh(i+2,j,k)+fh(i+3,j,k))
+
+     elseif(i+2 <= imax .and. i-2 >= imin)then
+!
+!              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
+!  fx(i) = ---------------------------------------------
+!                             12 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+     elseif(i+1 <= imax .and. i-1 >= imin)then
+!
+!              - f(i-1) + f(i+1)
+!  fx(i) = --------------------------------
+!                     2 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfx(i,j,k)*d2dx*(-fh(i-1,j,k)+fh(i+1,j,k))
+
+! set imax and imin 0
+    endif
+
+! y direction   
+     if(Sfy(i,j,k) >= ZEO .and. j+4 <= jmax .and. j-2 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                   &
+                  Sfy(i,j,k)*d60dy*(TWO*fh(i,j-2,k)-F24*fh(i,j-1,k)-F35*fh(i,j,k)+F80*fh(i,j+1,k) &
+                                   -F30*fh(i,j+2,k)+EIT*fh(i,j+3,k)-    fh(i,j+4,k))
+     elseif(Sfy(i,j,k) >= ZEO .and. j+5 <= jmax .and. j-1 >= jmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                        &
+                  Sfy(i,j,k)*d60dy*(-F10*fh(i,j-1,k)-F77*fh(i,j  ,k)+F150*fh(i,j+1,k)-F100*fh(i,j+2,k) &
+                                    +F50*fh(i,j+3,k)-F15*fh(i,j+4,k)+ TWO*fh(i,j+5,k))
+
+     elseif(Sfy(i,j,k) <= ZEO .and. j-4 >= jmin .and. j+2 <= jmax)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                   &
+                  Sfy(i,j,k)*d60dy*(TWO*fh(i,j+2,k)-F24*fh(i,j+1,k)-F35*fh(i,j,k)+F80*fh(i,j-1,k) &
+                                   -F30*fh(i,j-2,k)+EIT*fh(i,j-3,k)-    fh(i,j-4,k))
+
+     elseif(Sfy(i,j,k) <= ZEO .and. j-5 >= jmin .and. j+1 <= jmax)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                        &
+                  Sfy(i,j,k)*d60dy*(-F10*fh(i,j+1,k)-F77*fh(i,j  ,k)+F150*fh(i,j-1,k)-F100*fh(i,j-2,k) &
+                                    +F50*fh(i,j-3,k)-F15*fh(i,j-4,k)+ TWO*fh(i,j-5,k))
+
+     elseif(j+3 <= jmax .and. j-3 >= jmin)then
+          
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                                         &
+                  Sfy(i,j,k)*d60dy*(-fh(i,j-3,k)+F9*fh(i,j-2,k)-F45*fh(i,j-1,k)+F45*fh(i,j+1,k)-F9*fh(i,j+2,k)+fh(i,j+3,k))
+
+     elseif(j+2 <= jmax .and. j-2 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+     elseif(j+1 <= jmax .and. j-1 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfy(i,j,k)*d2dy*(-fh(i,j-1,k)+fh(i,j+1,k))
+! set jmin and jmax 0
+     endif
+!! z direction   
+     if(Sfz(i,j,k) >= ZEO .and. k+4 <= kmax .and. k-2 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                   &
+                  Sfz(i,j,k)*d60dz*(TWO*fh(i,j,k-2)-F24*fh(i,j,k-1)-F35*fh(i,j,k)+F80*fh(i,j,k+1) &
+                                   -F30*fh(i,j,k+2)+EIT*fh(i,j,k+3)-    fh(i,j,k+4))
+     elseif(Sfz(i,j,k) >= ZEO .and. k+5 <= kmax .and. k-1 >= kmin)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                        &
+                  Sfz(i,j,k)*d60dz*(-F10*fh(i,j,k-1)-F77*fh(i,j,k  )+F150*fh(i,j,k+1)-F100*fh(i,j,k+2) &
+                                    +F50*fh(i,j,k+3)-F15*fh(i,j,k+4)+ TWO*fh(i,j,k+5))
+
+     elseif(Sfz(i,j,k) <= ZEO .and. k-4 >= kmin .and. k+2 <= kmax)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                   &
+                  Sfz(i,j,k)*d60dz*(TWO*fh(i,j,k+2)-F24*fh(i,j,k+1)-F35*fh(i,j,k)+F80*fh(i,j,k-1) &
+                                   -F30*fh(i,j,k-2)+EIT*fh(i,j,k-3)-    fh(i,j,k-4))
+
+     elseif(Sfz(i,j,k) <= ZEO .and. k-5 >= kmin .and. k+1 <= kmax)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                        &
+                  Sfz(i,j,k)*d60dz*(-F10*fh(i,j,k+1)-F77*fh(i,j,k  )+F150*fh(i,j,k-1)-F100*fh(i,j,k-2) &
+                                    +F50*fh(i,j,k-3)-F15*fh(i,j,k-4)+ TWO*fh(i,j,k-5))
+     
+     elseif(k+3 <= kmax .and. k-3 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                                         &
+                  Sfz(i,j,k)*d60dz*(-fh(i,j,k-3)+F9*fh(i,j,k-2)-F45*fh(i,j,k-1)+F45*fh(i,j,k+1)-F9*fh(i,j,k+2)+fh(i,j,k+3))
+
+     elseif(k+2 <= kmax .and. k-2 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+     elseif(k+1 <= kmax .and. k-1 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+Sfz(i,j,k)*d2dz*(-fh(i,j,k-1)+fh(i,j,k+1))
+! set kmin and kmax 0
+     endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+  end subroutine lopsided
+
+#elif (ghost_width == 5)
+! eighth order code
+!-----------------------------------------------------------------------------
+! PRD 77, 024034 (2008)
+! Compute advection terms in right hand sides of field equations
+!        v [ - 5 f(i-3v) + 60 f(i-2v) - 420 f(i-v) - 378 f(i) + 1050 f(i+v) - 420 f(i+2v) + 140 f(i+3v) - 30 f(i+4v) + 3 f(i+5v)]
+! D f = --------------------------------------------------------------------------------------------------------------------------
+!  i                                                             840 dx           
+!
+! where
+!
+!        i
+!      |B |
+! v = -----
+!        i
+!       B
+!
+!-----------------------------------------------------------------------------
+subroutine lopsided(ex,X,Y,Z,f,f_rhs,Sfx,Sfy,Sfz,Symmetry,SoA)
+  implicit none
+
+!~~~~~~> Input parameters:
+
+  integer, intent(in)  :: ex(1:3),Symmetry
+  real*8,  intent(in)  :: X(1:ex(1)),Y(1:ex(2)),Z(1:ex(3))
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(in)   :: f,Sfx,Sfy,Sfz
+
+  real*8,dimension(ex(1),ex(2),ex(3)),intent(inout):: f_rhs
+  real*8,dimension(3),intent(in) ::SoA
+
+!~~~~~~> local variables:
+
+  real*8,dimension(-4:ex(1),-4:ex(2),-4:ex(3))   :: fh
+  integer :: imin,jmin,kmin,imax,jmax,kmax,i,j,k
+  real*8 :: dX,dY,dZ
+  real*8 :: d840dx,d840dy,d840dz,d60dx,d60dy,d60dz,d12dx,d12dy,d12dz,d2dx,d2dy,d2dz
+  real*8,  parameter :: ZEO=0.d0,ONE=1.d0, F60=6.d1
+  real*8,  parameter :: TWO=2.d0,F30=3.d1,EIT=8.d0
+  real*8,  parameter ::  F9=9.d0,F45=4.5d1,F12=1.2d1,F140=1.4d2,THR=3.d0
+  real*8,  parameter :: F840=8.4d2,F5=5.d0,F420=4.2d2,F378=3.78d2,F1050=1.05d3
+  real*8,  parameter :: F32=3.2d1,F168=1.68d2,F672=6.72d2
+  integer, parameter :: NO_SYMM = 0, EQ_SYMM = 1, OCTANT = 2
+
+  dX = X(2)-X(1)
+  dY = Y(2)-Y(1)
+  dZ = Z(2)-Z(1)
+
+  d840dx = ONE/F840/dX
+  d840dy = ONE/F840/dY
+  d840dz = ONE/F840/dZ
+
+  d60dx = ONE/F60/dX
+  d60dy = ONE/F60/dY
+  d60dz = ONE/F60/dZ
+
+  d12dx = ONE/F12/dX
+  d12dy = ONE/F12/dY
+  d12dz = ONE/F12/dZ
+
+  d2dx = ONE/TWO/dX
+  d2dy = ONE/TWO/dY
+  d2dz = ONE/TWO/dZ
+
+  imax = ex(1)
+  jmax = ex(2)
+  kmax = ex(3)
+
+  imin = 1
+  jmin = 1
+  kmin = 1
+  if(Symmetry > NO_SYMM .and. dabs(Z(1)) < dZ) kmin = -4
+  if(Symmetry > EQ_SYMM .and. dabs(X(1)) < dX) imin = -4
+  if(Symmetry > EQ_SYMM .and. dabs(Y(1)) < dY) jmin = -4
+
+  call symmetry_bd(5,ex,f,fh,SoA)
+
+! upper bound set ex-1 only for efficiency, 
+! the loop body will set ex 0 also
+  do k=1,ex(3)-1
+  do j=1,ex(2)-1
+  do i=1,ex(1)-1
+! x direction   
+    if(Sfx(i,j,k) >= ZEO .and. i+5 <= imax .and. i-3 >= imin)then
+!        v [ - 5 f(i-3v) + 60 f(i-2v) - 420 f(i-v) - 378 f(i) + 1050 f(i+v) - 420 f(i+2v) + 140 f(i+3v) - 30 f(i+4v) + 3 f(i+5v)]
+! D f = --------------------------------------------------------------------------------------------------------------------------
+!  i                                                             840 dx    
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                         &
+                  Sfx(i,j,k)*d840dx*(-F5*fh(i-3,j,k)+F60 *fh(i-2,j,k)-F420*fh(i-1,j,k)-F378*fh(i  ,j,k) &
+                                  +F1050*fh(i+1,j,k)-F420*fh(i+2,j,k)+F140*fh(i+3,j,k)-F30 *fh(i+4,j,k)+THR*fh(i+5,j,k))
+
+    elseif(Sfx(i,j,k) <= ZEO .and. i-5 >= imin .and. i+3 <= imax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                          &
+                  Sfx(i,j,k)*d840dx*(-F5*fh(i+3,j,k)+F60 *fh(i+2,j,k)-F420*fh(i+1,j,k)-F378*fh(i   ,j,k) &
+                                  +F1050*fh(i-1,j,k)-F420*fh(i-2,j,k)+F140*fh(i-3,j,k)- F30*fh(i-4,j,k)+THR*fh(i-5,j,k))
+
+    elseif(i+4 <= imax .and. i-4 >= imin)then
+!           3 f(i-4) - 32 f(i-3) + 168 f(i-2) - 672 f(i-1) + 672 f(i+1) - 168 f(i+2) + 32 f(i+3) - 3 f(i+4)
+!  fx(i) = -------------------------------------------------------------------------------------------------
+!                                                        840 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                              &
+                  Sfx(i,j,k)*d840dx*( THR*fh(i-4,j,k)-F32 *fh(i-3,j,k)+F168*fh(i-2,j,k)-F672*fh(i-1,j,k)+    &
+                                     F672*fh(i+1,j,k)-F168*fh(i+2,j,k)+F32 *fh(i+3,j,k)-THR *fh(i+4,j,k))
+
+     elseif(i+3 <= imax .and. i-3 >= imin)then
+!           - f(i-3) + 9 f(i-2) - 45 f(i-1) + 45 f(i+1) - 9 f(i+2) + f(i+3)
+!  fx(i) = -----------------------------------------------------------------
+!                                        60 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                              &
+                  Sfx(i,j,k)*d60dx*(-fh(i-3,j,k)+F9*fh(i-2,j,k)-F45*fh(i-1,j,k)+F45*fh(i+1,j,k)-F9*fh(i+2,j,k)+fh(i+3,j,k))
+
+     elseif(i+2 <= imax .and. i-2 >= imin)then
+!
+!              f(i-2) - 8 f(i-1) + 8 f(i+1) - f(i+2)
+!  fx(i) = ---------------------------------------------
+!                             12 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                           &
+                  Sfx(i,j,k)*d12dx*(fh(i-2,j,k)-EIT*fh(i-1,j,k)+EIT*fh(i+1,j,k)-fh(i+2,j,k))
+
+     elseif(i+1 <= imax .and. i-1 >= imin)then
+!
+!              - f(i-1) + f(i+1)
+!  fx(i) = --------------------------------
+!                     2 dx
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfx(i,j,k)*d2dx*(-fh(i-1,j,k)+fh(i+1,j,k))
+
+! set imax and imin 0
+    endif
+
+! y direction   
+    if(Sfy(i,j,k) >= ZEO .and. j+5 <= jmax .and. j-3 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                         &
+                  Sfy(i,j,k)*d840dy*(-F5*fh(i,j-3,k)+F60 *fh(i,j-2,k)-F420*fh(i,j-1,k)-F378*fh(i,j  ,k) &
+                                  +F1050*fh(i,j+1,k)-F420*fh(i,j+2,k)+F140*fh(i,j+3,k)-F30 *fh(i,j+4,k)+THR*fh(i,j+5,k))
+
+    elseif(Sfy(i,j,k) <= ZEO .and. j-5 >= jmin .and. j+3 <= jmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                         &
+                  Sfy(i,j,k)*d840dy*(-F5*fh(i,j+3,k)+F60 *fh(i,j+2,k)-F420*fh(i,j+1,k)-F378*fh(i,j  ,k) &
+                                  +F1050*fh(i,j-1,k)-F420*fh(i,j-2,k)+F140*fh(i,j-3,k)- F30*fh(i,j-4,k)+THR*fh(i,j-5,k))
+
+    elseif(j+4 <= jmax .and. j-4 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                              &
+                  Sfy(i,j,k)*d840dy*( THR*fh(i,j-4,k)-F32 *fh(i,j-3,k)+F168*fh(i,j-2,k)-F672*fh(i,j-1,k)+    &
+                                     F672*fh(i,j+1,k)-F168*fh(i,j+2,k)+F32 *fh(i,j+3,k)-THR *fh(i,j+4,k))
+
+     elseif(j+3 <= jmax .and. j-3 >= jmin)then
+          
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                                         &
+                  Sfy(i,j,k)*d60dy*(-fh(i,j-3,k)+F9*fh(i,j-2,k)-F45*fh(i,j-1,k)+F45*fh(i,j+1,k)-F9*fh(i,j+2,k)+fh(i,j+3,k))
+
+     elseif(j+2 <= jmax .and. j-2 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfy(i,j,k)*d12dy*(fh(i,j-2,k)-EIT*fh(i,j-1,k)+EIT*fh(i,j+1,k)-fh(i,j+2,k))
+
+     elseif(j+1 <= jmax .and. j-1 >= jmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k) + Sfy(i,j,k)*d2dy*(-fh(i,j-1,k)+fh(i,j+1,k))
+! set jmin and jmax 0
+     endif
+!! z direction   
+    if(Sfz(i,j,k) >= ZEO .and. k+5 <= kmax .and. k-3 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                         &
+                  Sfz(i,j,k)*d840dz*(-F5*fh(i,j,k-3)+F60 *fh(i,j,k-2)-F420*fh(i,j,k-1)-F378*fh(i,j,k  ) &
+                                  +F1050*fh(i,j,k+1)-F420*fh(i,j,k+2)+F140*fh(i,j,k+3)-F30 *fh(i,j,k+4)+THR*fh(i,j,k+5))
+
+    elseif(Sfz(i,j,k) <= ZEO .and. k-5 >= kmin .and. k+3 <= kmax)then
+     f_rhs(i,j,k)=f_rhs(i,j,k)-                                                                         &
+                  Sfz(i,j,k)*d840dz*(-F5*fh(i,j,k+3)+F60 *fh(i,j,k+2)-F420*fh(i,j,k+1)-F378*fh(i,j,k  ) &
+                                  +F1050*fh(i,j,k-1)-F420*fh(i,j,k-2)+F140*fh(i,j,k-3)- F30*fh(i,j,k-4)+THR*fh(i,j,k-5))
+
+    elseif(k+4 <= kmax .and. k-4 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                              &
+                  Sfz(i,j,k)*d840dz*( THR*fh(i,j,k-4)-F32 *fh(i,j,k-3)+F168*fh(i,j,k-2)-F672*fh(i,j,k-1)+    &
+                                     F672*fh(i,j,k+1)-F168*fh(i,j,k+2)+F32 *fh(i,j,k+3)-THR *fh(i,j,k+4))
+     
+     elseif(k+3 <= kmax .and. k-3 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                                                         &
+                  Sfz(i,j,k)*d60dz*(-fh(i,j,k-3)+F9*fh(i,j,k-2)-F45*fh(i,j,k-1)+F45*fh(i,j,k+1)-F9*fh(i,j,k+2)+fh(i,j,k+3))
+
+     elseif(k+2 <= kmax .and. k-2 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+                                                            &
+                  Sfz(i,j,k)*d12dz*(fh(i,j,k-2)-EIT*fh(i,j,k-1)+EIT*fh(i,j,k+1)-fh(i,j,k+2))
+
+     elseif(k+1 <= kmax .and. k-1 >= kmin)then
+
+     f_rhs(i,j,k)=f_rhs(i,j,k)+Sfz(i,j,k)*d2dz*(-fh(i,j,k-1)+fh(i,j,k+1))
+! set kmin and kmax 0
+     endif
+
+  enddo
+  enddo
+  enddo
+
+  return
+
+  end subroutine lopsided
+
+#endif  
--- a/parallel_plot_helper.py
+++ b/parallel_plot_helper.py
@@ -0,0 +1,29 @@
+import multiprocessing
+
+def run_plot_task(task):
+    """Execute a single plotting task.
+    
+    Parameters
+    ----------
+    task : tuple
+        A tuple of (function, args_tuple) where function is a callable
+        plotting function and args_tuple contains its arguments.
+    """
+    func, args = task
+    return func(*args)
+
+
+def run_plot_tasks_parallel(plot_tasks):
+    """Execute a list of independent plotting tasks in parallel.
+
+    Uses the 'fork' context to create worker processes so that the main
+    script is NOT re-imported/re-executed in child processes.
+
+    Parameters
+    ----------
+    plot_tasks : list of tuples
+        Each element is (function, args_tuple).
+    """
+    ctx = multiprocessing.get_context('fork')
+    with ctx.Pool() as pool:
+        pool.map(run_plot_task, plot_tasks)
--- a/plot_GW_strain_amplitude_xiaoqu.py
+++ b/plot_GW_strain_amplitude_xiaoqu.py
@@ -11,6 +11,8 @@
 import numpy                               ## numpy for array operations
 import scipy                               ## scipy for interpolation and signal processing
 import math
+import matplotlib
+matplotlib.use('Agg')                      ## use non-interactive backend for multiprocessing safety
 import matplotlib.pyplot    as     plt     ## matplotlib for plotting
 import os                                  ## os for system/file operations

--- a/plot_binary_data.py
+++ b/plot_binary_data.py
@@ -8,16 +8,23 @@
 ##
 #################################################

+## Restrict OpenMP to one thread per process so that running
+## many workers in parallel does not create an O(workers * BLAS_threads)
+## thread explosion.  The variable MUST be set before numpy/scipy
+## are imported, because the BLAS library reads them only at load time.
+import os
+os.environ.setdefault("OMP_NUM_THREADS",        "1")
+
 import numpy
 import scipy
+import matplotlib
+matplotlib.use('Agg')                      ## use non-interactive backend for multiprocessing safety
 import matplotlib.pyplot    as     plt
 from   matplotlib.colors    import LogNorm
 from   mpl_toolkits.mplot3d import Axes3D
 ## import torch
 import AMSS_NCKU_Input      as input_data

-import os
-

 #########################################################################################

@@ -192,3 +199,19 @@ def get_data_xy( Rmin, Rmax, n, data0, time, figure_title, figure_outdir ):

 ####################################################################################

+
+####################################################################################
+## Allow this module to be run as a standalone script so that each
+## binary-data plot can be executed in a fresh subprocess whose BLAS
+## environment variables (set above) take effect before numpy loads.
+##
+## Usage:  python3 plot_binary_data.py <filename> <binary_outdir> <figure_outdir>
+####################################################################################
+
+if __name__ == '__main__':
+    import sys
+    if len(sys.argv) != 4:
+        print(f"Usage: {sys.argv[0]} <filename> <binary_outdir> <figure_outdir>")
+        sys.exit(1)
+    plot_binary_data(sys.argv[1], sys.argv[2], sys.argv[3])
+
--- a/plot_xiaoqu.py
+++ b/plot_xiaoqu.py
@@ -8,6 +8,8 @@
 #################################################

 import numpy                               ## numpy for array operations
+import matplotlib
+matplotlib.use('Agg')                      ## use non-interactive backend for multiprocessing safety
 import matplotlib.pyplot    as     plt     ## matplotlib for plotting
 from   mpl_toolkits.mplot3d import Axes3D  ## needed for 3D plots
 import glob
@@ -15,6 +17,9 @@ import os                                  ## operating system utilities

 import plot_binary_data
 import AMSS_NCKU_Input as input_data
+import subprocess
+import sys
+import multiprocessing

 # plt.rcParams['text.usetex'] = True  ## enable LaTeX fonts in plots

@@ -50,10 +55,40 @@ def generate_binary_data_plot( binary_outdir, figure_outdir ):
        file_list.append(x)
        print(x)

-    ## Plot each file in the list
+    ## Plot each file in parallel using subprocesses.
+    ## Each subprocess is a fresh Python process where the BLAS thread-count
+    ## environment variables (set at the top of plot_binary_data.py) take
+    ## effect before numpy is imported.  This avoids the thread explosion
+    ## that occurs when multiprocessing.Pool with 'fork' context inherits
+    ## already-initialized multi-threaded BLAS from the parent.
+    script = os.path.join( os.path.dirname(__file__), "plot_binary_data.py" )
+    max_workers = min( multiprocessing.cpu_count(), len(file_list) ) if file_list else 0
+
+    running = []
+    failed  = []
    for filename in file_list:
        print(filename)
-        plot_binary_data.plot_binary_data(filename, binary_outdir, figure_outdir)
+        proc = subprocess.Popen(
+            [sys.executable, script, filename, binary_outdir, figure_outdir],
+        )
+        running.append( (proc, filename) )
+        ## Keep at most max_workers subprocesses active at a time
+        if len(running) >= max_workers:
+            p, fn = running.pop(0)
+            p.wait()
+            if p.returncode != 0:
+                failed.append(fn)
+
+    ## Wait for all remaining subprocesses to finish
+    for p, fn in running:
+        p.wait()
+        if p.returncode != 0:
+            failed.append(fn)
+
+    if failed:
+        print( " WARNING: the following binary data plots failed:" )
+        for fn in failed:
+            print( "   ", fn )

    print(                        )
    print( " Binary Data Plot Has been Finished " )
Author	SHA1	Message	Date
ianchb	8b68b5d782	fixup! Fix load explosion: use subprocess for binary data plots to avoid thread conflict * Seems we don't have to set so many variables, `OMP_NUM_THREADS` is enough. Test: Annotate the code for setting other environment variables. It runs normally.	2026-02-09 23:00:17 +08:00
ianchb	dd2443c926	Fix load explosion: use subprocess for binary data plots to avoid thread conflict Co-authored-by: copilot-swe-agent[bot] <198982749+copilot@users.noreply.github.com>	2026-02-09 21:40:27 +08:00
ianchb	2d7ba5c60c	[2/2] Implement multiprocessing-based parallel plotting	2026-02-09 21:36:45 +08:00
ianchb	4777cad4ed	[1/2] Implement multiprocessing-based parallel plotting	2026-02-09 15:13:18 +08:00
ianchb	afd4006da2	Cache GSL in SyncPlan and apply async Sync to Z4c_class Major optimization: Pre-build grid segment lists (GSLs) once per Step() call via SyncPreparePlan(), then reuse them across all 4 RK4 substep SyncBegin calls via SyncBeginWithPlan(). This eliminates the O(cpusize * blocks^2) GSL rebuild cost that was incurred on every ghost zone exchange. Applied async SyncBegin/SyncEnd overlap pattern to Z4c_class.C (ABEtype==2, the default configuration), which was still using blocking Parallel::Sync. Both the regular and CPBC variants of Z4c Step() are now optimized. Co-authored-by: copilot-swe-agent[bot] <198982749+copilot@users.noreply.github.com>	2026-02-08 16:46:44 +08:00
copilot-swe-agent[bot]	a918dc103e	Add SyncBegin/SyncEnd to Parallel for MPI communication-computation overlap Split the blocking Parallel::Sync into async SyncBegin (initiates local copy + MPI_Isend/Irecv) and SyncEnd (MPI_Waitall + unpack). This allows overlapping MPI ghost zone exchange with error checking and Shell patch computation. Modified Step() in bssn_class.C for both PSTR==0 and PSTR==1/2/3 versions to start Sync before error checks, overlapping the MPI_Allreduce with the ongoing ghost zone transfers. Co-authored-by: copilot-swe-agent[bot] <198982749+copilot@users.noreply.github.com>	2026-02-08 16:19:13 +08:00
copilot-swe-agent[bot]	38c2c30186	Merge lopsided advection + kodis dissipation to share symmetry_bd buffer Add lopsided_kodis subroutine in lopsidediff.f90 that combines upwind advection (lopsided) and Kreiss-Oliger dissipation (kodis) into one function sharing a single fh buffer from symmetry_bd. This eliminates 27 redundant full-grid copies per RHS evaluation (108 per timestep). For gxx/gyy/gzz variables: kodis stencil coefficients sum to zero (1-6+15-20+15-6+1=0), so using gxx(=dxx+1) instead of dxx for the dissipation buffer is mathematically exact. Update bssn_rhs.f90 to use the merged lopsided_kodis calls. Co-authored-by: ianchb <45872450+ianchb@users.noreply.github.com>	2026-02-08 15:42:44 +08:00
ianchb	b8e41b2b39	Only enable OpenMP for TwoPunctures	2026-02-08 13:00:37 +08:00
ianchb	133e4f13a2	Use OpenMP's parallel for with schedule(dynamic,1)	2026-02-07 19:48:24 +08:00
ianchb	914c4f4791	Optimize memory allocation in JFD_times_dv This should reduce the pressure on the memory allocator, indirectly improving caching behavior. Co-authored-by: copilot-swe-agent[bot] <198982749+copilot@users.noreply.github.com>	2026-02-07 15:55:45 +08:00