Hi,
I have a .ksh which in-turn is calling another script in loop based on the number of files one by one. Need to change the process so that it can run in parallel based on the number of threads and pick up files to process. For example if there are 5 threads and 30 files to process hence first 5 threads should process first 5 files in parallel and as soon as one thread completes it picks up next file to process.
Below is my current program:
# global script variables
typeset -i programReturnCode # the script exit code
# Named constants used within the script
typeset -i normalReturnCode=0; readonly normalReturnCode
typeset -i fatalReturnCode=255; readonly fatalReturnCode
typeset -r logDir=${LOG_DIR:-"."} # default to current directory if log variable not set
typeset -r jobName="reimediinjector"
typeset -r scriptDir=$RETAIL_HOME/reim-batch/batch
typeset -r fileMask="*.txt"
typeset -r invFileList="$(basename $0 .ksh).$$.list"
typeset invFile # holds a individual invoices from the file list
typeset -r dateFormat="+%m/%d/%Y %H:%M:%S"
typeset -r fileTimeStamp="+%Y%m%d%H%M%S"
typeset -r programTimeStamp="$(date $fileTimeStamp)"
typeset -r programLogFile="$(basename $0 .ksh).$programTimeStamp.log"
typeset -r programID="$0 PID:$$"
#---------------------------------------------------------------------------------
# Function Name: PrintLog
# Purpose : Function used to print the input string to the program log file
#---------------------------------------------------------------------------------
function PrintLog
{
print -- "$@" >> $logDir/$programLogFile
} # PrintLog
#---------------------------------------------------------------------------------
# Function Name: Usage
# Purpose : Defines how the program should be invoked
#---------------------------------------------------------------------------------
function Usage
{
echo "USAGE: $(basename $0) <db_connect> <input_directory> <archive_directory> <reject_directory>
<db_connect> The reim batch alias to be used to invoke ReIM batch processes
<input_directory> The path where the invoice files to be processed are present
<archive_directory> The path where the successfully processed invoice files are moved.
<reject_directory> The path where the failed invoice files are moved."
} #Usage
#---------------------------------------------------------------------------------
# Function Name: ValidateParms
# Purpose : Validates the script parameters
#---------------------------------------------------------------------------------
function ValidateParms
{
typeset -i li_return_code
typeset ls_log_msg
# assume valid parms until proven otherwise
li_return_code=$normalReturnCode
# validate batch job access
if [[ ! -f $scriptDir/$jobName ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot find the batch job: $jobName"
li_return_code=$fatalReturnCode
elif [[ ! -x $scriptDir/$jobName ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot execute the batch job: $jobName"
li_return_code=$fatalReturnCode
fi
# validate input directory access
if [[ ! -d $inputDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot find data directory: $inputDir"
li_return_code=$fatalReturnCode
elif [[ ! -r $inputDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot read from data directory: $inputDir"
li_return_code=$fatalReturnCode
fi
# validate archive directory access
if [[ ! -d $archiveDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot find archive directory: $archiveDir"
li_return_code=$fatalReturnCode
elif [[ ! -w $archiveDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot write to archive directory: $archiveDir"
li_return_code=$fatalReturnCode
fi
# validate reject directory access
if [[ ! -d $rejDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot find reject directory: $rejDir"
li_return_code=$fatalReturnCode
elif [[ ! -w $rejDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot write to reject directory: $rejDir"
li_return_code=$fatalReturnCode
fi
# validate log directory access
if [[ ! -d $logDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot find log directory: $logDir"
li_return_code=$fatalReturnCode
elif [[ ! -w $logDir ]]
then
ls_log_msg="$ls_log_msg\nFAILURE: Cannot write to log directory: $logDir"
li_return_code=$fatalReturnCode
fi
# log any validation errors
if [[ ! -z $ls_log_msg ]]
then
print $ls_log_msg
PrintLog "$ls_log_msg"
fi
return $li_return_code
} # ValidateParms
#---------------------------------------------------------------------------------
# Function Name: ProcessFileList
# Purpose : Function that will parse all the files available and invoke the base injector for each file
#---------------------------------------------------------------------------------
function ProcessFileList
{
typeset -i li_return_code
# initialize parameters
li_return_code=$normalReturnCode
# process files one by one
while read invFileParam
do
invFile=$(basename $invFileParam)
# call the base injector for each invoice file present
PrintLog "./$jobName $db_connect $invFileParam $rejDir/$invFile.reg"
$scriptDir/$jobName $db_connect $invFileParam $rejDir/$invFile.reg
jobExitCode=$?
if [[ $jobExitCode -eq 0 ]]
then
PrintLog "\nChecking existence of reject file $invFile.reg in reject directory: $rejDir"
rejectFileCount=`ls $rejDir/$invFile.reg | wc -l`
if [[ $rejectFileCount -eq 0 ]]
then
PrintLog "\n No rejected files were detected under $rejDir"
# archive the file with a .done and timestamp extension
mv $invFileParam $archiveDir/$invFile.done.$programTimeStamp
else
PrintLog "\n Renaming the rejected file $rejDir/$invFile.error.$programTimeStamp"
# rename the file with a .error abd timestamp extension and move to rejected directory
mv $invFileParam $rejDir/$invFile.error.$programTimeStamp
fi
PrintLog "\nFile $invFile exit code: $jobExitCode \tstop $(date "$dateFormat")"
else
# failure, reject file
mv $invFileParam $rejDir/$invFile.reg
PrintLog "$File $invFile exit code: $jobExitCode \tstop $(date "$dateFormat")"
li_return_code=$fatalReturnCode
fi
done < $invFileList # while job parm
# wait for all jobs to finish
wait
return $li_return_code
} # ProcessFileList
#--------------------------------------
# Logs the script termination and then kills its spawned jobs and itself
#--------------------------------------
function TerminateScript
{
# log termination
PrintLog "\nSignal received, terminating script\n$programID stop $(date "$dateFormat") with return code 15"
# clean work files
rm -f $invFileList 2> /dev/null
# terminate all the processes having a process group ID equal to the process group ID of the sender
kill -KILL 0
# exit script with a terminated signal
exit 15
} # TerminateScript
#--------------------------------------
# Signal trap
# Terminate script on signals 1) HUP, 2) INT, 3) QUIT, or 15) TERM
#--------------------------------------
trap 'TerminateScript' 1 2 3 15
#######################################
# Program Execution starts here
# If the required number of command line parms not present,
# notify user how to call script and exit
if [[ $1 = "?" ]] || (($# < 4))
then
Usage
exit $fatalReturnCode
fi
# initialize variables
programReturnCode=$normalReturnCode
typeset -r db_connect=$1
typeset -r inputDir=$2
typeset -r archiveDir=$3
typeset -r rejDir=$4
# initialize program log
PrintLog "\n$programID started at $(date "$dateFormat")"
PrintLog "\tDatabase Connection \t: $db_connect"
PrintLog "\tInput Directory \t: $inputDir"
PrintLog "\tArchive Directory \t: $archiveDir"
PrintLog "\tReject Directory \t: $rejDir"
# validate script paramenters
ValidateParms
programReturnCode=$?
# process data file list
if ((programReturnCode == normalReturnCode))
then
# build data file list
ls $inputDir/$fileMask 2> /dev/null > $invFileList
# if we have files to process..
if [[ -s $invFileList ]]
then
# Process list with restart flag
PrintLog "\nProcessing data file list"
ProcessFileList
programReturnCode=$?
else
PrintLog "\nThere are no invoices to process"
fi
fi # data files
# clean work files
rm -f $invFileList 2> /dev/null
# conclude program log
PrintLog "\n$programID finished at $(date "$dateFormat") with return code $programReturnCode"
exit $programReturnCode
