#!/usr/bin/mawk -We
# *********************************************************************
# nblparser: experimental NoSQL Brokering Language (NBL) interpreter.
#
# Copyright (c) 2003,2006 Carlo Strozzi
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; version 2 dated June, 1991.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA.
#
# *********************************************************************

BEGIN {
  NULL = ""; FS = OFS = "\t"; tmptable_cmd = "tmptable"; pfx = ""

  # Get local settings.
  nosql_install = ENVIRON["NOSQL_INSTALL"]
  stdout = ENVIRON["NOSQL_STDOUT"]
  stderr = ENVIRON["NOSQL_STDERR"]

  # Set default values if necessary.
  if (nosql_install == NULL) nosql_install = "/usr/local/nosql"
  if (stdout == NULL) stdout = "/dev/stdout"
  if (stderr == NULL) stderr = "/dev/stderr"

  # Process command-line options and arguments.

  while (ARGV[++i] != NULL) {
    if (ARGV[i] == "-i" || ARGV[i] == "--input") i_file = ARGV[++i]
    else if (ARGV[i] == "-o" || ARGV[i] == "--output") o_file = ARGV[++i]
    else if (ARGV[i] == "-d" || ARGV[i] == "--delete") {
       tmptable_cmd = "tmptable --delete " ARGV[++i]
    }
    else if (ARGV[i] == "-U" || ARGV[i] == "--unsafe") unsafe = 1
    else if (ARGV[i] == "-D" || ARGV[i] == "--directory") dir = ARGV[++i]
    else if (ARGV[i] == "-P" || ARGV[i] == "--prefix") pfx = ARGV[++i]
    else if (ARGV[i] == "-n" || ARGV[i] == "--no-hide") nohide = 1
    else if (ARGV[i] == "-h") {
       system("grep -v '^#' " nosql_install "/help/nblparser.txt")
       exit(rc=1)
    }
    else if (ARGV[i] == "--show-copying") {
       system("cat " nosql_install "/doc/COPYING")
       exit(rc=1)
    }
    else if (ARGV[i] == "--show-warranty") {
       system("cat " nosql_install "/doc/WARRANTY")
       exit(rc=1)
    }
    else if (s_files == "") s_files = ARGV[i]
    else s_names = ARGV[i]
  }

  ARGC = 1					# Fix argv[]

  if (o_file == NULL) o_file = stdout
  if (i_file != NULL) { ARGV[1] = i_file; ARGC = 2 }

  if (s_files == "") {
     print "usage: nblparser [options] file-schema [field-schema]" > stderr
     exit(rc=1)
  } else if (s_files !~ /^\// && dir != "") s_files = dir "/" s_files

  # Load file-schema into memory
  i=0
  if ((getline < s_files) <= 0) {
     print "nblparser: could not read schema table '"s_files"'" > stderr
     exit(rc=1)
  }

  # Remove SOH markers.
  gsub(/\001/, "")

  # Load the column position array.
  i=0; while (++i <= NF) p[$i] = i

  # check for required 'Table' field.
  if (!p[pfx "Table"]) {
     print "nblparser: missing " pfx "'Table' column in schema table '"s_files"'" > stderr
     exit(rc=1)
  }

  # 'Path' defaults to 'Table' if not present.
  if (!p[pfx "Path"]) p[pfx "Path"] = p[pfx "Table"]

  # load s_files body.
  i=0; while (getline schema[++i] < s_files > 0);
  schema[0] = i				# save array length

  close(s_files);

  # Load field-schema into memory.

  if (s_names != "") {
    if (s_names !~ /^\// && dir != "") s_names = dir "/" s_names
    i=0
    if ((getline < s_names) <= 0) {
       print "nblparser: could not read schema table '"s_names"'" > stderr
       exit(rc=1)
    }

    # Remove SOH markers.
    gsub(/\001/, "")

    # Load the column position array.
    i=0; while (++i <= NF) fp[$i] = i

    # check for required fields.
    if (!fp[pfx "Column"] || !fp[pfx "Flags"]) {
       print "nblparser: required column(s) missing in schema table '"s_names"'" > stderr
       exit(rc=1)
    }

    # load s_names body.
    i=0
    while ((getline fschema[++i] < s_names) > 0) {
      split(fschema[i],a,"\t")
      value = tolower(a[fp[pfx "Flags"]])

      # Hide actual column names, not their symbolic names!
      if (value ~ /h/) {
	 if (schema_attrs[a[fp[pfx "Column"]]] != "")
		hide = hide " " schema_attrs[a[fp[pfx "Column"]]]
	 else hide = hide " " schema_columns[a[fp[pfx "Column"]]]
      }

      # work-out key column names.

      if (value ~ /[Kk]/) {
	 if (keylist[a[fp[pfx "Table"]]] != "")
	     keylist[a[fp[pfx "Table"]]] = keylist[a[fp[pfx "Table"]]] ","
	 keylist[a[fp[pfx "Table"]]] = \
			keylist[a[fp[pfx "Table"]]] a[fp[pfx "Column"]]
      }
    }
    close(s_names);
    fschema[0] = i			# save array length
  }

  FS = " "				# set default FS

  print "#!/bin/sh\nset -e" > o_file
  if (dir != "") print "cd " dir > o_file
}

# Read input NBL statements.

# skip comments and empty lines.
/^[ \t]*(#|$)/ { next }

{ sub(/^[ \t]+/,"") }			# Trim leading blanks and tabs.

# Add new NBL verbs as needed. Note: at the moment only the "ldap",
# "columns" and "@columns" handlers use column real names if defined
# in the "schema:Aliasof" field, while all other handlers only use
# the "schema:Column" values and assume that those are the actual
# column names, not aliases. Column hiding on output also works on
# "Aliasof" if available.

$1 == "use"		{ handle_use();			next }
$1 == "setvar"		{ handle_setvar();		next }
$1 == "read"		{ handle_read();		next }
$1 == "expr"		{ handle_expr("expression");	next }
$1 == "expression"	{ handle_expr($1);		next }
$1 == "ldap"		{ handle_expr($1);		next }
$1 == "select"		{ handle_expr($1);		next }
$1 == "compute"		{ handle_expr($1);		next }
$1 == "columns"		{ handle_column();		next }
$1 == "@columns"	{ handle_column("revert");	next }
$1 == "join"		{ handle_join();		next }
$1 == "@join"		{ handle_join("outer");		next }
$1 == "order-by"	{ handle_orderby();		next }
$1 == "@order-by"	{ handle_orderby("revert");	next }
$1 == "unique-by"	{ handle_uniqueby();		next }
$1 == "@unique-by"	{ handle_uniqueby("revert");	next }
$1 == "remember-as"	{ handle_remember();		next }
$1 == "totals"		{ handle_totals();		next }
$1 == "@totals"		{ handle_totals("currency");	next }
$1 == "format"		{ handle_format();		next }
$1 == "@format"		{ handle_format("list");	next }
$1 == "system"		{ handle_system();		next }
$1 == "labels"		{ handle_labels();		next }
$1 == "sizelimit"	{ size_limit = $2 + 1;		next }
$1 == "end"		{ exit(0) }

# ending stuff.
END {
  if (rc) exit(rc)

  # debug
  #if (schema_row[1] != "") {
  #   printf("#") > o_file
  #   i=0
  #   while (schema_row[++i] != "") printf(" %s", schema_row[i]) > o_file
  #   printf("\n") > o_file
  #}

  # Skip hiding command in a number of cases, included when the
  # latest shell statement ends with ')', as a result of the
  # 'remember-as' statement.

  if (!nohide && hide != "" && sh_cmd[sh_cmd[0]] !~ /\)$/)
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nnotcolumn " hide

  if (size_limit)
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nhead -" size_limit

  if (labels != "")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nsetnames " labels

  i=0
  while (sh_cmd[++i] != "") print sh_cmd[i] > o_file
}

function check_cmd(handler) {

  if (schema_row[p[pfx "Table"]] == "") {
     print "nblparser: NBL '" handler "': no active table, use 'read' first" > stderr
     exit(rc=1)
  }
}

function ldap_to_awk(q,		name,expr) {
  sub(/^\(/,"",q); sub(/\)$/,"",q)
  name = substr(q,1,index(q,"=") - 1)
  expr = substr(q,index(q,"=") + 1)
  gsub(/[]\\\$()\[\|\^\?\.]/,"\\\\&",expr)	# escape AWK special chars.
  gsub(/\*/,".*",expr)
  expr = "/^" expr "$/"

  name = tolower(name)

  # Use case-insensitive LDAP names to fetch the actual
  # case-sensitive NoSQL field names.

  if (schema_attrs[name] != "") name = schema_attrs[name]
  else name = schema_columns[name]

  # Ignore nonexisting columns, or 'awktable' will output the entire
  # row if they appare in an expression.

  if (name == "") return ""

  # LDAP queries are case-insensitive.
  return "tolower($" name ") ~ " tolower(expr)
}

# NBL handlers

function handle_setvar(			i, target) {

  if (NF != 2 && NF != 3) {
     print "nblparser: NBL usage: set name [value]" > stderr
     exit(rc=1)
  }

  if ($2 !~ /^[A-Za-z_][A-Za-z0-9_]*$/) {
     print "nblparser: NBL 'set': bad name in assignment" > stderr
     exit(rc=1)
  }

  # The value part must be acceptable as a directory or file name,
  # with no other path components.
  if ($3 != "" && ($3 ~ /^\.\.$/ || $3 !~ /^[-_.,=:+A-Za-z0-9]*$/)) {
     print "nblparser: NBL 'set': bad value in assignment" > stderr
     exit(rc=1)
  }

  # Replace value in schema "path" field.
  target = "\\$\\[" $2 "\\]"
  while (++i <= schema[0]) gsub(target,$3,schema[i])
}


function handle_read(		i,a,k,file,partial,cmd,edits) {

  if (schema_row[p[pfx "Table"]] != "") {
     print "nblparser: NBL 'read': cannot open multiple tables" > stderr
     exit(rc=1)
  }

  if (NF < 2 || NF > 3) {
     print "nblparser: NBL usage: read table [key]" > stderr
     exit(rc=1)
  }

  delete schema_row
  delete schema_tables
  delete schema_columns
  delete schema_attrs

  # a partial key string must begin with a "@" sign.
  partial = sub(/^@/,"",$3)

  if ($2 !~ /^\$?[A-Za-z_][A-Za-z0-9_.]*$/) {
     print "nblparser: NBL 'read': bad table or variable name" > stderr
     exit(rc=1)
  }

  # check whether variable name.

  if ($2 ~ /^\$/) {

     # variable names are more restrictive than table names.

     if ($2 !~ /^\$[A-Za-z_][A-Za-z0-9_]*$/) {
	print "nblparser: NBL 'read': bad variable name" > stderr
	exit(rc=1)
     }

     schema_row[p[pfx "Table"]] = $2	# needed by other handlers

     # we assume that "set -e" is used in the output sh(1) script,
     # so that if test(1) fails the script terminates.

     file = "\"" $2 "\""
     cmd = "test -f " file "; "
  }

  if ($3 != "") {
     if ($3 !~ /^[A-Za-z0-9]+$/) {
     	print "nblparser: NBL 'read': bad key specified" > stderr
     	exit(rc=1)
     }
  }

  # lookup target table in schema, unless "$table"

  if (file == "") {

    while (++i <= schema[0]) {
       split(schema[i],a,"\t")
       # Ignore duplicated table names in schema array.
       if (a[p[pfx "Table"]] == $2 && schema_tables[$2] == "") {
	  schema_tables[$2] = $2
	  split(schema[i],schema_row,"\t")
	  if (dir != "" && schema_row[p[pfx "Path"]] !~ /^\//)
		schema_row[p[pfx "Path"]] = \
				dir "/" schema_row[p[pfx "Path"]]
	  break				# bail-out if table found
       }
    }

    if (schema_row[p[pfx "Table"]] == "") {
       print "nblparser: NBL 'read': unknown table '" $2 "'" > stderr
       exit(rc=1)
    }

    # Load target table's column names.
    i=0
    while (++i <= schema[0]) {
       split(schema[i],a,"\t")
       if (a[p[pfx "Table"]] == $2) {

	  schema_columns[tolower(a[p[pfx "Column"]])] = a[p[pfx "Column"]]

	  if (a[p[pfx "Aliasof"]] != "")
	     schema_attrs[tolower(a[p[pfx "Column"]])] = a[p[pfx "Aliasof"]]
	  else schema_attrs[tolower(a[p[pfx "Column"]])] = a[p[pfx "Column"]]

	  ldap_all = ldap_all " " schema_columns[tolower(a[p[pfx "Column"]])]
       }
    }

    if (schema_row[p[pfx "Path"]] ~ /\$\[/) {
       print "nblparser: NBL 'read': variable tokens in path name, use 'set' first" > stderr
       exit(rc=1)
    }

    file = schema_row[p[pfx "Path"]]
  }
	
  if ($3 != "") {
     post_filter = $3
     if (partial) {
	sh_cmd[++sh_cmd[0]] = cmd "keysearch -p '" $3 "' " file
	gsub(/[]\\\$()\[\|\^\*\?\.]/,"\\\\&",post_filter)
	post_filter = "awk '$1 ~ /^(\\001|" post_filter ")/'"
     }
     else {
	sh_cmd[++sh_cmd[0]] = cmd "keysearch '" $3 "' " file
	gsub(/\\/,"\\\\&",post_filter)
	post_filter = "awk '$1 ~ /^\\001/ || $1 == \"" post_filter "\"'"
     }
  }
  else sh_cmd[++sh_cmd[0]] = cmd "cat " file

  # Set default edit file name.
  if (schema_row[p[pfx "Edit"]] == "")
		edit_file = schema_row[p[pfx "Table"]] "-edits"
  else edit_file = schema_row[p[pfx "Edit"]]

  # Test whether an edit file exists for this table.

  if (getline edits < edit_file >= 0) {

     close(edit_file)

     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nupdtable"

     key_field = keylist[schema_row[p[pfx "Table"]]]

     if (key_field != "")
	sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " --key-columns " key_field

     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " " edit_file

     if (post_filter != "")
	sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\n" post_filter
  }
}


function handle_expr(what,		a,i,j,regexp,q,tmp) {

  check_cmd(what)

  if (NF < 2) {
     print "nblparser: NBL usage: " what " statements" > stderr
     exit(rc=1)
  }

  regexp = "^" what "[ \t]+"
  sub(regexp,"")

  if (what == "select") what = "awktable -r -H --"
  else if (what == "compute") what = "awktable -c -H --"

  # Forbid AWK's dangerous instructions, if necessary. The following
  # is a rough check, which may be stricter than necessary in some cases,
  # but better to be safe than sorry.

  if (!unsafe) {

     i = split($0,a,/[^a-z]+/)

     while (++j <= i) {
       if (a[j] ~ /^(printf?|getline|system)$/) {
	  print "nblparser: unsafe AWK instruction specified" > stderr
	  exit(rc=1)
       }
     }
  }

  ldap_suffix = ""

  if (what == "ldap") {

     if ($1 ~ /[()]/ || $2 !~ /[()]/) {
	print "nblparser: too few arguments in LDAP expression" > stderr
	exit(rc=1)
     }

     schema_columns["dn"] = "dn" 		# Mandatory.

     ldap_suffix = $1

     $0 = $2

     what = "expression"

     # Strip trailing and middle blanks/tabs.
     sub(/\)[ \t]*$/,")"); gsub(/\)[ \t]*\(/,")(")

     # Detect (&(name1=value1) (&(name2=value2) [(&(...))] ))
     # and turn it into (&(name1=value1) [(name2=value2) ...])
     # 
     # This is hackish, we'll see.

     if (gsub(/\&\(/,"")) {
	gsub(/\(+/,"(")
	gsub(/\)+/,")")
	$0 = "(&" $0 ")"
     }

     # Detect (name=value)
     if ($0 ~ /^\([A-Za-z]+[A-Za-z0-9_]*=[^()]+\)$/) q = ldap_to_awk($0)

     # Detect (&(name1=value1) [(name2=value2) ...])
     else if ($0 ~ /^\(\&(\([A-Za-z]+[A-Za-z0-9_]*=[^()]+\))+\)$/) {
       sub(/^\(\&\(/,""); sub(/\)\)$/,"")
       i = split($0,a,/\)\(/)
       for (j=1; j<=i; j++) {
	   if ((tmp=ldap_to_awk(a[j])) != "") {
	      if (j > 1) q = q " && "
	      q = q "(" tmp ")"
	   }
       }
     }

     # Detect (|(name1=value1) [(name2=value2) ...])
     else if ($0 ~ /^\(\|(\([A-Za-z]+[A-Za-z0-9_]*=[^()]+\))+\)$/) {
       sub(/^\(\|\(/,""); sub(/\)\)$/,"")
       i = split($0,a,/\)\(/)
       for (j=1; j<=i; j++) {
	   if ((tmp=ldap_to_awk(a[j])) != "") {
	      if (j > 1) q = q " || "
	      q = q "(" tmp ")"
	   }
       }
     }

     if (q == "") {
	print "nblparser: unsupported LDAP expression in input" > stderr
	exit(rc=1)
     }

     if (schema_attrs["cn"] != "")
        $0 = "{$dn=\"cn=\" $" schema_attrs["cn"] " \"," ldap_suffix "\"}" q
     else
        $0 = "{$dn=\"cn=\" $" schema_columns["cn"] " \"," ldap_suffix "\"}" q
  }

  # escape sh(1) special characters in statements.

  gsub(/\\/, "&&")
  gsub(/\$/, "\\$")
  gsub(/`/, "\\`")
  gsub(/"/,"\\\"")

  if (what == "expression") {
     if (ldap_suffix != "") what = "nltable --key dn"
     what = what " |\nawktable -H --"
     $0 = $0 "{print}"
  }
  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\n" what " \"" $0 "\""

  if (ldap_suffix != "") {
     $0 = "columns " ldap_all
     handle_column()
  }
}


function handle_column(revert,			i,a,cols) {

  if (revert != "") {
     check_cmd("@columns")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\ngetcolumn -r"
  }
  else {
     check_cmd("columns")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\ngetcolumn"
  }


  if (NF < 2) {
     print "nblparser: NBL usage: columns col [col ...]" > stderr
     exit(rc=1)
  }

  sub(/^@?columns[ \t]+/,"")

  # handle LDAP attribute specs in the form '* attribute [attribute ...]'.
  if (ldap_suffix != "") sub(/.*\*/,ldap_all)

  split($0,a)
  while (a[++i] != "") {
     if (a[i] !~ /^[A-Za-z_][A-Za-z0-9_]*$/) {
	print "nblparser: NBL 'columns': bad column names(s) specified" > stderr
	exit(rc=1)
     }
  }

  # Get actual column names from symbolic names.
  i = split($0,a,/[ \t]+/)
  $0 = ""
  for (j=1; j<=i; j++) {
      a[j] = tolower(a[j])
      if (schema_attrs[a[j]] != "") cols = cols " " schema_attrs[a[j]]
      else cols = cols " " schema_columns[a[j]]

      $0 = $0 " " schema_columns[a[j]]
  }

  if (ldap_suffix != "") {

     # Remove any misplaced references to 'dn'
     # from both 'cols' and '$0'.

     gsub(/[ \t]+[Dd][Nn][ \t]+/," ",cols)
     sub(/^[Dd][Nn][ \t]+/,"",cols)
     sub(/[ \t][Dd][Nn]$/,"",cols)

     # $0 already contains actual column names.
     gsub(/[ \t]+[Dd][Nn][ \t]+/," ")
     sub(/^[Dd][Nn][ \t]+/,"")
     sub(/[ \t][Dd][Nn]$/,"")

     # Remove any misplaced references to 'cn'.
     # from both 'cols' and '$0'.
     #gsub(/[ \t]+[Cc][Nn][ \t]+/," ",cols)
     #sub(/^[Cc][Nn][ \t]+/,"",cols)
     #sub(/[ \t][Cc][Nn]$/,"",cols)

     # 'Cn' may have been aliased in the table schema,
     # so make sure we are referring to its actual name.

     re = "[ \\t]+" schema_attrs["cn"] "[ \\t]+"; gsub(re," ",cols)
     re = "^" schema_attrs["cn"] "[ \\t]+"; gsub(re,"",cols)
     re = "[ \\t]" schema_attrs["cn"] "$"; sub(re,"",cols)

     # $0 already contains actual column names.
     gsub(/[ \t]+[Cc][Nn][ \t]+/," ")
     sub(/^[Cc][Nn][ \t]+/,"")
     sub(/[ \t][Cc][Nn]$/,"")

     # Output 'dn' and 'cn' first, in due order.
     cols = " dn " schema_columns["cn"] cols
     $0 = "dn cn" $0
  }

  $0 = "labels " $0
  handle_labels()

  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] cols
}


function handle_join(outer,		i,a,b,k,\
					join_row,tbl,cmd,file) {

  if (outer != "") check_cmd("@join")
  else check_cmd("join")

  if (NF < 2 || NF > 4) {
     print "nblparser: NBL usage: [@]join table1 [table2 [columns]]" > stderr
     exit(rc=1)
  }

  # set default arguments

  if ($3 == "") $3 = "-"
  if ($4 == "") $4 = "-"

  # either one or the other table, but not both, must be on stdin.

  if (($2 == "-" && $3 == "-") || ($2 != "-" && $3 != "-")) {
     print "nblparser: NBL 'join': one (and only one) of the input tables must be on stdin" > stderr
     exit(rc=1)
  }

  if (($2 != "-" && $2 !~ /^@?\$?[A-Za-z_][A-Za-z0-9_]*$/) || \
      ($3 != "-" && $3 !~ /^@?\$?[A-Za-z_][A-Za-z0-9_]*$/)) {
     print "nblparser: NBL 'join': bad table or variable name" > stderr
     exit(rc=1)
  }

  cmd = "jointable"
  if (outer != "") cmd = cmd " --all"

  if ($2 != "-") tbl = $2
  else tbl = $3

  # check whether variable name.

  if (tbl ~ /^\$/) {

     # we assume that "set -e" is used in the output sh(1) script,
     # so that if test(1) fails the script terminates.

     file = "\"" tbl "\""
  }

  # lookup target table in schema, unless "$table"

  if (file == "") {

    while (++i <= schema[0]) {
       split(schema[i],a,"\t")
       if (a[p[pfx "Table"]] == tbl) {
	  split(schema[i],join_row,"\t")
	  break					# bail-out if table found
       }
    }

    if (join_row[p[pfx "Table"]] == "") {
       print "nblparser: NBL 'join': unknown table '" tbl "'" > stderr
       exit(rc=1)
    }

    if (join_row[p[pfx "Path"]] ~ /\$\[/) {
       print "nblparser: NBL 'join': variable tokens in file name, use 'set' first" > stderr
       exit(rc=1)
    }

    file = join_row[p[pfx "Path"]]

    # Try to infer the key column(s) from the file name, if possible.

    jlist = file

    # Handle both table and index file names (the index first!). Note
    # that _k and _x were choosen because no real column name can begin
    # with an uderscore, so there's no risk of ambiguities. Note also
    # that we need to strip everything up to _x first, as in index
    # files the actual key columns are those that come after _x, and
    # they may not necessarily be the same as the key columns of the
    # main table. That is, given the main table 'table._k.col1.col2',
    # it is quite possible to have an index file name like this:
    # 'table._k.col1.col2._x.col3.col4.col5

    if (sub(/.*\._x\./,"",jlist) || sub(/.*\._k\./,"",jlist)) {
       gsub(/\./,",",jlist)
       sub(/-.*$/,"",jlist)		# remove possible "-suffix".
    }
    else jlist = ""
  }

  # join column(s) from both tables, if specified.

  if ($4 != "" && $4 != "-") {
     if ($4 !~ /^@?([A-Za-z][A-Za-z0-9_]*)(,[A-Za-z][A-Za-z0-9_]*)*$/) {
     	print "nblparser: NBL 'join': bad column name(s) specified" > stderr
     	exit(rc=1)
     }

     jlist = $4
  }

  if (jlist != "") cmd = cmd " --column " jlist

  if ($2 != "-") cmd = cmd " " file " -"
  else cmd = cmd " - " file

  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\n" cmd
}


function handle_orderby(revert,		i,cmd) {

  if (revert != "") check_cmd("@order-by")
  else check_cmd("order-by")

  cmd = "sorttable"
  if (revert != "") cmd = cmd " -r"

  if ($2 == "-") NF=1			# allow "-" to mean "all columns"

  for (i=2; i<=NF; i++) {
      if ($i !~ /^[A-Za-z][A-Za-z0-9_]+(:[Mbdfinr])?$/) {
	 print "nblparser: NBL 'order-by': bad column name specified" > stderr
	 exit(rc=1)
      }
      cmd = cmd " " $i
  }
  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\n" cmd
}

function handle_uniqueby(revert,		i,cmd) {

  if (revert != "") check_cmd("@unique-by")
  else check_cmd("unique-by")

  cmd = "sorttable -u"
  if (revert != "") cmd = cmd " -r"

  if ($2 == "-") NF=1			# allow "-" to mean "all columns"

  for (i=2; i<=NF; i++) {
      if ($i !~ /^[A-Za-z][A-Za-z0-9_]+(:[Mbdfinr])?$/) {
	 print "nblparser: NBL 'unique-by': bad column name specified" > stderr
	 exit(rc=1)
      }
      cmd = cmd " " $i
  }
  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\n" cmd
}


function handle_use(			i,j,a,t,tbl,file,row) {

  if (schema_row[p[pfx "Table"]] != "") {
     print "nblparser: the 'use' statement, if used, must come first" > stderr
     exit(rc=1)
  }

  if (locklist != "") {
     print "nblparser: multiple 'use' statements are not allowed" > stderr
     exit(rc=1)
  }

  if (NF < 2) {
     print "nblparser: NBL usage: use table [table ...]" > stderr
     exit(rc=1)
  }

  sub(/^use[ \t]+/,"")

  j = split($0,tbl)
  
  # for each listed table.
  for (t=1; t<=j; t++) {

     if (tbl[t] !~ /^\$?[A-Za-z_][A-Za-z0-9_]*$/) {
        print "nblparser: NBL 'use': bad table or variable name" > stderr
        exit(rc=1)
     }

     file = ""

     # check whether it is a variable name.

     if (tbl[t] ~ /^\$/) {

        #schema_row[p[pfx "Table"]] = tbl[t]	# needed by other handlers

        # we assume that "set -e" is used in the output sh(1) script,
        # so that if test(1) fails the script terminates.

        file = "\"" tbl[t] "\""
	print "test -f " file > o_file
     }

     # lookup target table in schema, unless "$table"

     if (file == "") {

       while (++i <= schema[0]) {
          split(schema[i],a,"\t")
          if (a[p[pfx "Table"]] == tbl[t]) {
	     split(schema[i],row,"\t")

	     # If an edit buffer is used, then lock that one instead
	     # of main table.

	     if (row[p[pfx "Edit"]] != "") file = row[p[pfx "Edit"]]
	     else file = row[p[pfx "Path"]]
	     if (dir != "" && file !~ /^\//) file = dir "/" file
	     break				# bail-out if table found
          }
       }

       if (file == "") {
          print "nblparser: NBL 'use': unknown table '" tbl[t] "'" > stderr
          exit(rc=1)
       }

       if (file ~ /\$\[/) {
          print "nblparser: NBL 'use': variable tokens in path name, use 'set' first" > stderr
          exit(rc=1)
       }

       locklist = locklist " " file ".lock"
     }
  }

  print "lockfile -r6 -l40 -s8 " locklist > o_file
  print "trap 'rm -f " locklist "' 0" > o_file
  print "trap 'exit 2' 1 2 3 15" > o_file
}


function handle_remember() {

  check_cmd("remember-as")

  if (NF != 2) {
     print "nblparser: NBL usage: remember-as name" > stderr
     exit(rc=1)
  }

  # accept only valid variable names. Lower-case is mandatory, not to
  # interfere with the usual shell environment variables.

  if ($2 !~ /^[a-z][a-z0-9_]*$/) {
     print "nblparser: NBL 'remember-as': bad name specified" > stderr
     exit(rc=1)
  }

  sh_cmd[sh_cmd[0]] = $2 "=$(\n" sh_cmd[sh_cmd[0]] " |\n" tmptable_cmd "\n)"

  delete schema_row		# a new 'read' is necessary from now on.
}


function handle_format(list) {

  if (!nohide && hide != "") {
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nnotcolumn " hide
     nohide = 1			# tell END{} to hide no more.
  }

  if (list != "") {
     check_cmd("@format")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\ntabletolist --justify"
  }
  else {
     check_cmd("format")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\nprtable"
  }

  delete schema_row		# a new 'read' is necessary from now on.
}


function handle_labels(			a,i) {

  if (NF < 2) {
     print "nblparser: NBL usage: labels label [label ...]" > stderr
     exit(rc=1)
  }

  sub(/^labels[ \t]+/,"")
  split($0,a)
  while (a[++i] != "") {
     if (a[i] !~ /^[A-Za-z_][A-Za-z0-9_]*$/) {
	print "nblparser: NBL 'labels': bad label names(s) specified" > stderr
	exit(rc=1)
     }
  }

  labels = $0
}


function handle_totals(currency,	i,a) {

  if (currency != "") {
     check_cmd("@totals")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\ntotaltable -c"
  }
  else {
     check_cmd("totals")
     sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |\ntotaltable"
  }

  sub(/^@?totals[ \t]*/,"")
  split($0,a)
  while (a[++i] != "") {
     if (a[i] !~ /^[A-Za-z_][A-Za-z0-9_]*$/) {
	print "nblparser: NBL 'totals': bad column names(s) specified" > stderr
	exit(rc=1)
     }
  }

  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " " $0
}


function handle_system() {

  if (!unsafe) {
     print "nblparser: NBL 'system': command not allowed" > stderr
     exit(rc=1)
  }

  if (NF < 2) {
     print "nblparser: NBL usage: system commands" > stderr
     exit(rc=1)
  }

  sub(/^system[ \t]*/,"")

  if (schema_row[p[pfx "Table"]] != "")
	sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] " |"
  sh_cmd[sh_cmd[0]] = sh_cmd[sh_cmd[0]] "\n" $0
}

# End of program
