From 5544b7a5a1b2db722c6da8b9063bb1cd86b28a71 Mon Sep 17 00:00:00 2001 From: zaihuaji Date: Fri, 4 Sep 2026 11:07:44 -0500 Subject: [PATCH] reconcile the legacy Pg*.py modules with their pg_*.py classes; bump version to 3.0.14 The module-level Pg*.py twins had drifted years behind the active class-based pg_*.py implementations, so downstream callers still importing the old names were silently running stale (and in places broken) logic. All nine legacy modules are now AST-equivalent to their active counterparts, which also picks up the accumulated bug fixes (Globus task waits, dssgrp-only user lookup, setuid chmod, psutil process scans, hashlib md5) and drops the retired HPSS, SLURM and UCAR People DB code paths. open_output/OUTPUT deliberately stays in PgOPT.py, since ten downstream files reference PgOPT.OUTPUT and it needs PGOPT['extlog']; it now also assigns PgLOG.OUTPUT so the ported PgLOG.pgexit() can still close it. Also fixes two defects that were present in both trees: endtime() split 'HH:MM:SS' on a literal 'T' and raised IndexError, and tosystem() declared logact=0 while the body kept the "if logact is None" idiom, so the intended LOGWRN default was dead. Co-Authored-By: Claude Opus 4.6 --- README.md | 2 +- pyproject.toml | 2 +- src/rda_python_common/PgCMD.py | 30 +- src/rda_python_common/PgDBI.py | 389 ++++++++---------------- src/rda_python_common/PgFile.py | 306 ++++++++----------- src/rda_python_common/PgLOG.py | 483 ++++++++++-------------------- src/rda_python_common/PgLock.py | 21 +- src/rda_python_common/PgOPT.py | 80 ++--- src/rda_python_common/PgSIG.py | 470 ++++++++++------------------- src/rda_python_common/PgSplit.py | 71 ++--- src/rda_python_common/PgUtil.py | 336 +++++++-------------- src/rda_python_common/__init__.py | 2 +- src/rda_python_common/pg_log.py | 4 +- src/rda_python_common/pg_util.py | 2 +- 14 files changed, 766 insertions(+), 1432 deletions(-) diff --git a/README.md b/README.md index fa593f7..91c52aa 100644 --- a/README.md +++ b/README.md @@ -165,7 +165,7 @@ PgLOG.pglog("hello", PgLOG.LOGWRN) python -c "import rda_python_common; print(rda_python_common.__version__)" ``` -You should see the installed version (currently `3.0.13`). If the import +You should see the installed version (currently `3.0.14`). If the import fails, double-check that the active Python environment is the one where you ran `pip install`. diff --git a/pyproject.toml b/pyproject.toml index 431365d..8cf8a03 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "rda_python_common" -version = "3.0.13" +version = "3.0.14" authors = [ { name="Zaihua Ji", email="zji@ucar.edu" }, ] diff --git a/src/rda_python_common/PgCMD.py b/src/rda_python_common/PgCMD.py index d8545a3..63344d8 100644 --- a/src/rda_python_common/PgCMD.py +++ b/src/rda_python_common/PgCMD.py @@ -133,26 +133,21 @@ def get_delay_options(doptions, cmd): # # find an existing dscheck record from the cached command argument; create and initialize one if not exist # -def init_dscheck(oindex, otype, cmd, dsid, action, workdir = None, specialist = None, doptions = None, logact = 0): - +def init_dscheck(oindex, otype, cmd, dsid, action, workdir=None, specialist=None, doptions=None, logact=0): cidx = 0 argv = PgLOG.argv_to_string(sys.argv[1:], 0, "Process in Delayed Mode") argextra = None - if not logact: logact = PgLOG.LGEREX if not workdir: workdir = os.getcwd() if not specialist: specialist = PgLOG.PGLOG['CURUID'] - (mcount, hosts) = get_delay_options(doptions, cmd) - if len(argv) > 100: argextra = argv[100:] argv = argv[0:100] - bck = PgLOG.PGLOG['BCKGRND'] PgLOG.PGLOG['BCKGRND'] = 0 cinfo = "{}-{}-Chk".format(PgLOG.PGLOG['HOSTNAME'], PgLOG.current_datetime()) - pgrec = get_dscheck(cmd, argv, workdir, specialist, argextra, logact) + pgrec = get_dscheck(cmd, argv, workdir, specialist, argextra, logact) if pgrec: # found existing dscheck record cidx = pgrec['cindex'] cmsg = "{}{}: {} batch process ".format(cinfo, cidx, get_command_info(pgrec)) @@ -171,10 +166,9 @@ def init_dscheck(oindex, otype, cmd, dsid, action, workdir = None, specialist = PgLOG.pglog("{}is {}".format(cmsg, ('Done' if pgrec['status'] == 'D' else 'Finished')), PgLOG.LOGWRN) PgLock.lock_dscheck(cidx, 0, logact) sys.exit(0) - if not cidx: # add new dscheck record record = {} - if hosts and re.match(r'^(ds\d|\d)\d\d.\d$', hosts): + if hosts and re.match(r'^\w\d\d\d\d\d\d$', hosts): PgLOG.pglog(hosts + ": Cannot pass DSID for hostname to submit batch process", PgLOG.LGEREX) if oindex: set_command_control(oindex, otype, cmd, logact) record['oindex'] = oindex @@ -219,7 +213,6 @@ def init_dscheck(oindex, otype, cmd, dsid, action, workdir = None, specialist = if pgrec['command'] == "dsrqst" and pgrec['oindex']: (record['fcount'], record['dcount'], record['size']) = get_dsrqst_counts(pgrec, logact) PgDBI.pgupdt("dscheck", record, DSCHK['chkcnd'], logact) - DSCHK['dcount'] = pgrec['dcount'] DSCHK['fcount'] = pgrec['fcount'] DSCHK['size'] = pgrec['size'] @@ -234,7 +227,6 @@ def init_dscheck(oindex, otype, cmd, dsid, action, workdir = None, specialist = if rhost != chost: pstr += "/{}<{}>".format(rhost, rpid) PgLOG.pglog("{}Starts {} ({})".format(cmsg, tstr, pstr), PgLOG.LOGWRN) PgLOG.PGLOG['BCKGRND'] = bck - return cidx # @@ -318,17 +310,17 @@ def get_partition_control(pgpart, pgrqst = None, pgctl = None, logact = 0): # # build the dynamic options # -def get_dynamic_options(cmd, oindex, otype): - - if oindex: cmd += " {}".format(oindex) - if otype: cmd += ' ' + otype +def get_dynamic_options(cmd, oindex, otype, hostname=None): + if oindex: cmd += " {}".format(oindex) + if otype: cmd += ' ' + otype + if hostname: cmd = "ssh {} {}".format(hostname, PgLOG.command_path(cmd)) ret = options = '' for loop in range(3): - ret = PgLOG.pgsystem(cmd, PgLOG.LOGWRN, 279) # 1+2+4+16+256 + ret = PgLOG.pgsystem(cmd, PgLOG.LOGWRN, 1299) # 1+2+16+256+1024 if loop < 2 and PgLOG.PGLOG['SYSERR'] and 'Connection timed out' in PgLOG.PGLOG['SYSERR']: time.sleep(PgSIG.PGSIG['ETIME']) else: - break + break if ret: ret = ret.strip() ms = re.match(r'^(-.+)/(-.+)$', ret) @@ -337,8 +329,8 @@ def get_dynamic_options(cmd, oindex, otype): elif re.match(r'^(-.+)$', ret): options = ret if not options: - if ret: PgLOG.PGLOG['SYSERR'] += ret - PgLOG.PGLOG['SYSERR'] += " for {}".format(cmd) + if ret: PgLOG.PGLOG['SYSERR'] = (PgLOG.PGLOG['SYSERR'] or '') + ret + PgLOG.PGLOG['SYSERR'] = (PgLOG.PGLOG['SYSERR'] or '') + " for {}".format(cmd) return options diff --git a/src/rda_python_common/PgDBI.py b/src/rda_python_common/PgDBI.py index aedf00f..86c2a48 100644 --- a/src/rda_python_common/PgDBI.py +++ b/src/rda_python_common/PgDBI.py @@ -74,6 +74,7 @@ def get_pgerror(pgerr): SYSDOWN = {} PGDBI = {} ADDTBLS = [] +USRWARN = 0 # 1 - reminder for incomplete user records given already PGSIGNS = ['!', '<', '>', '<>'] CHCODE = 1042 @@ -89,8 +90,6 @@ def get_pgerror(pgerr): DBNAMES = { 'ivaddb' : 'ivaddb', 'cntldb' : 'ivaddb', - 'ivaddb1' : 'ivaddb', - 'cntldb1' : 'ivaddb', 'cdmsdb' : 'ivaddb', 'ispddb' : 'ispddb', 'obsua' : 'upadb', @@ -259,8 +258,7 @@ def get_dbname(scname): # set connection for viewing database information # def view_dbinfo(scname = None, lnname = None, pwname = None): - - return view_scinfo(get_dbname(scname), scname, lnname, pwname) + view_scinfo(get_dbname(scname), scname, lnname, pwname) # # set connection for viewing database/schema information @@ -276,9 +274,8 @@ def view_scinfo(dbname = None, scname = None, lnname = None, pwname = None): # set connection for given scname # def set_dbname(scname = None, lnname = None, pwname = None, dbhost = None, dbport = None, socket = None): - if not scname: scname = PGDBI['DEFSC'] - return set_scname(get_dbname(scname), scname, lnname, pwname, dbhost, dbport, socket) + set_scname(get_dbname(scname), scname, lnname, pwname, dbhost, dbport, socket) # # set connection for given database & schema names @@ -320,9 +317,7 @@ def set_scname(dbname = None, scname = None, lnname = None, pwname = None, dbhos # start a database transaction and exit if fails # def starttran(): - global curtran - if curtran == 1: endtran() # try to end previous transaction if not pgdb: pgconnect(0, 0, False) @@ -345,7 +340,6 @@ def starttran(): # end a transaction with changes committed and exit if fails # def endtran(autocommit = True): - global curtran if curtran and pgdb: if not pgdb.closed: pgdb.commit() @@ -356,7 +350,6 @@ def endtran(autocommit = True): # end a transaction without changes committed and exit inside if fails # def aborttran(autocommit = True): - global curtran if curtran and pgdb: if not pgdb.closed: pgdb.rollback() @@ -366,19 +359,17 @@ def aborttran(autocommit = True): # # record error message to dscheck record and clean the lock # -def record_dscheck_error(errmsg, logact = PGDBI['EXITLG']): - +def record_dscheck_error(errmsg, logact = None): + if logact is None: logact = PGDBI['EXITLG'] check = PgLOG.PGLOG['DSCHECK'] chkcnd = check['chkcnd'] if 'chkcnd' in check else "cindex = {}".format(check['cindex']) dflags = check['dflags'] if 'dflags' in check else '' if PgLOG.PGLOG['NOQUIT']: PgLOG.PGLOG['NOQUIT'] = 0 - pgrec = pgget("dscheck", "mcount, tcount, lockhost, pid", chkcnd, logact) if not pgrec: return 0 if not pgrec['pid'] and not pgrec['lockhost']: return 0 (chost, cpid) = PgLOG.current_process_info() if pgrec['pid'] != cpid or pgrec['lockhost'] != chost: return 0 - # update dscheck record only if it is still locked by the current process record = {} record['chktime'] = int(time.time()) @@ -390,19 +381,17 @@ def record_dscheck_error(errmsg, logact = PGDBI['EXITLG']): record['mcount'] = pgrec['mcount'] + 1 else: record['dflags'] = '' - if errmsg: errmsg = PgLOG.break_long_string(errmsg, 512, None, 50, None, 50, 25) if pgrec['tcount'] > 1: errmsg = "Try {}: {}".format(pgrec['tcount'], errmsg) record['errmsg'] = errmsg - return pgupdt("dscheck", record, chkcnd, logact) # # local function to log query error # -def qelog(dberror, sleep, sqlstr, vals, pgcnt, logact = PGDBI['ERRLOG']): - +def qelog(dberror, sleep, sqlstr, vals, pgcnt, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] retry = " Sleep {}(sec) & ".format(sleep) if sleep else " " if sqlstr: if sqlstr.find("Retry ") == 0: @@ -416,22 +405,22 @@ def qelog(dberror, sleep, sqlstr, vals, pgcnt, logact = PGDBI['ERRLOG']): sqlstr = retry + sqlstr else: sqlstr = '' - if vals: sqlstr += " with values: " + str(vals) - if dberror: sqlstr = "{}\n{}".format(dberror, sqlstr) if logact&PgLOG.EXITLG and PgLOG.PGLOG['DSCHECK']: record_dscheck_error(sqlstr, logact) PgLOG.pglog(sqlstr, logact) - if sleep: time.sleep(sleep) - + if sleep: time.sleep(sleep) return PgLOG.FAILURE # if not exit in PgLOG.pglog() # # try to add a new table according the table not exist error # def try_add_table(dberror, logact): - - ms = re.match(r'^42P01 ERROR: relation "(.+)" does not exist', dberror) + # The caller (check_dberror) only invokes this after pgcode == '42P01', so + # match just the relation message. psycopg2's pgerror includes the + # 'ERROR: ' severity prefix while psycopg3's message_primary does not, so + # anchoring on that prefix only matched under psycopg2. + ms = re.search(r'relation "(.+)" does not exist', dberror) if ms: tname = ms.group(1) add_new_table(tname, logact = logact) @@ -480,10 +469,9 @@ def valid_table(tname, pre = None, suf = None, logact = 0): # # local function to log query error # -def check_dberror(pgerr, pgcnt, sqlstr, ary, logact = PGDBI['ERRLOG']): - +def check_dberror(pgerr, pgcnt, sqlstr, ary, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] ret = PgLOG.FAILURE - pgcode = get_pgcode(pgerr) pgerror = get_pgerror(pgerr) dberror = "{} {}".format(pgcode, pgerror) if pgcode and pgerror else str(pgerr) @@ -499,7 +487,7 @@ def check_dberror(pgerr, pgcnt, sqlstr, ary, logact = PGDBI['ERRLOG']): qelog(dberror, 0, "Retry Connecting", ary, pgcnt, PgLOG.LOGWRN) pgconnect(1, pgcnt + 1) return (PgLOG.FAILURE if not pgdb else PgLOG.SUCCESS) - elif re.match(r'^55', pgcode): # try to lock again + elif pgcode.startswith('55'): # try to lock again qelog(dberror, 10, "Retry Locking", ary, pgcnt, PgLOG.LOGWRN) return PgLOG.SUCCESS elif pgcode == '25P02': # try to add table @@ -510,7 +498,6 @@ def check_dberror(pgerr, pgcnt, sqlstr, ary, logact = PGDBI['ERRLOG']): qelog(dberror, 0, "Retry after adding a table", ary, pgcnt, PgLOG.LOGWRN) try_add_table(dberror, logact) return PgLOG.SUCCESS - if logact&PgLOG.DOLOCK and pgcode and re.match(r'^55\w\w\w$', pgcode): logact &= ~PgLOG.EXITLG # no exit for lock error elif pgcnt > PgLOG.PGLOG['DBRETRY']: @@ -520,32 +507,21 @@ def check_dberror(pgerr, pgcnt, sqlstr, ary, logact = PGDBI['ERRLOG']): # # return hash reference to postgresql batch mode command and output file name # -def pgbatch(sqlfile, foreground = 0): - -# if(PGDBI['VWHOST'] and PGDBI['VWHOME'] and -# PGDBI['DBSHOST'] == PGDBI['VWSHOST'] and PGDBI['SCNAME'] == PGDBI['VWNAME']): -# slave = "/{}/{}.slave".format(PGDBI['VWHOME'], PGDBI['VWHOST']) -# if not op.exists(slave): default_scname() - +def pgbatch(sqlfile, foreground=0): dbhost = 'localhost' if PGDBI['DBSHOST'] == PgLOG.PGLOG['HOSTNAME'] else PGDBI['DBHOST'] options = "-h {} -p {}".format(dbhost, PGDBI['DBPORT']) - pwname = get_pgpass_password() - os.environ['PGPASSWORD'] = pwname + os.environ['PGPASSWORD'] = get_pgpass_password() options += " -U {} {}".format(PGDBI['LNNAME'], PGDBI['DBNAME']) - if not sqlfile: return options - if foreground: - batch = "psql {} < {} |".format(options, sqlfile) + return "psql {} < {} |".format(options, sqlfile) + batch = {} + batch['out'] = sqlfile + if re.search(r'\.sql$', batch['out']): + batch['out'] = re.sub(r'\.sql$', '.out', batch['out']) else: - batch['out'] = sqlfile - if re.search(r'\.sql$', batch['out']): - batch['out'] = re.sub(r'\.sql$', '.out', batch['out']) - else: - batch['out'] += ".out" - - batch['cmd'] = "psql {} < {} > {} 2>&1".format(options, sqlfile , batch['out']) - + batch['out'] += ".out" + batch['cmd'] = "psql {} < {} > {} 2>&1".format(options, sqlfile, batch['out']) return batch # @@ -553,17 +529,14 @@ def pgbatch(sqlfile, foreground = 0): # force connect if connect > 0 # def pgconnect(reconnect = 0, pgcnt = 0, autocommit = True): - global pgdb - if pgdb: if reconnect and not pgdb.closed: return pgdb # no need reconnect elif reconnect: reconnect = 0 # initial connection - while True: - config = {'dbname' : PGDBI['DBNAME'], - 'user' : PGDBI['LNNAME']} + config = {'dbname': PGDBI['DBNAME'], + 'user': PGDBI['LNNAME']} if PGDBI['DBSHOST'] == PgLOG.PGLOG['HOSTNAME']: config['host'] = 'localhost' else: @@ -617,7 +590,6 @@ def pgcursor(): # disconnect to dssdb database # def pgdisconnect(stopit = 1): - global pgdb if pgdb: if stopit: pgdb.close() @@ -628,8 +600,8 @@ def pgdisconnect(stopit = 1): # and default values as values # the whole table information is cached to a hash array with table names as keys # -def pgtable(tablename, logact = PGDBI['ERRLOG']): - +def pgtable(tablename, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] if tablename in TABLES: return TABLES[tablename].copy() # cached already intms = r'^(smallint||bigint|integer)$' fields = "column_name col, data_type typ, is_nullable nil, column_default def" @@ -644,7 +616,6 @@ def pgtable(tablename, logact = PGDBI['ERRLOG']): else: return PgLOG.pglog(tablename + ": Table not exists", logact) pgcnt += 1 - pgdefs = {} for i in range(cnt): name = pgrecs['col'][i] @@ -662,21 +633,19 @@ def pgtable(tablename, logact = PGDBI['ERRLOG']): else: dflt = '' pgdefs[name] = dflt - TABLES[tablename] = pgdefs.copy() return pgdefs # # get sequence field name for given table name # -def pgsequence(tablename, logact = PGDBI['ERRLOG']): - +def pgsequence(tablename, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] if tablename in SEQUENCES: return SEQUENCES[tablename] # cached already condition = table_condition(tablename) + " AND column_default LIKE 'nextval(%'" pgrec = pgget('information_schema.columns', 'column_name', condition, logact) seqname = pgrec['column_name'] if pgrec else None SEQUENCES[tablename] = seqname - return seqname # @@ -754,17 +723,15 @@ def prepare_defaults(tablename, records, logact = 0): # record: hash reference with keys as field names and hash values as field values # return PgLOG.SUCCESS or PgLOG.FAILURE # -def pgadd(tablename, record, logact = PGDBI['ERRLOG'], getid = None): - +def pgadd(tablename, record, logact = None, getid = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not record: return PgLOG.pglog("Nothing adds to " + tablename, logact) if logact&PgLOG.DODFLT: prepare_default(tablename, record, logact) if logact&PgLOG.AUTOID and not getid: getid = pgsequence(tablename, logact) sqlstr = prepare_insert(tablename, list(record), True, getid) values = tuple(record.values()) - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "Insert: " + str(values)) - ret = acnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -782,14 +749,12 @@ def pgadd(tablename, record, logact = PGDBI['ERRLOG'], getid = None): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgadd: 1 record added to " + tablename + ", return " + str(ret)) if(logact&PgLOG.ENDLCK): endtran() elif curtran: curtran += acnt if curtran > PGDBI['MTRANS']: starttran() - return ret # @@ -798,28 +763,24 @@ def pgadd(tablename, record, logact = PGDBI['ERRLOG'], getid = None): # records: dict with field names as keys and each value is a list of field values # return PgLOG.SUCCESS or PgLOG.FAILURE # -def pgmadd(tablename, records, logact = PGDBI['ERRLOG'], getid = None): - +def pgmadd(tablename, records, logact = None, getid = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not records: return PgLOG.pglog("Nothing to insert to table " + tablename, logact) if logact&PgLOG.DODFLT: prepare_defaults(tablename, records, logact) if logact&PgLOG.AUTOID and not getid: getid = pgsequence(tablename, logact) multi = True if getid else False - sqlstr = prepare_insert(tablename, list(records), multi, getid) - + sqlstr = prepare_insert(tablename, list(records), multi, getid) v = records.values() values = list(zip(*v)) cntrow = len(values) ids = [] if getid else None - if PgLOG.PGLOG['DBGLEVEL']: for row in values: PgLOG.pgdbg(1000, "Insert: " + str(row)) - count = pgcnt = 0 while True: pgcur = pgcursor() if not pgcur: return PgLOG.FAILURE - if getid: while count < cntrow: record = values[count] @@ -838,16 +799,13 @@ def pgmadd(tablename, records, logact = PGDBI['ERRLOG'], getid = None): if not check_dberror(pgerr, pgcnt, sqlstr, values[0], logact): return PgLOG.FAILURE if count >= cntrow: break pgcnt += 1 - pgcur.close() if(PgLOG.PGLOG['DBGLEVEL']): PgLOG.pgdbg(1000, "pgmadd: {} of {} record(s) added to {}".format(count, cntrow, tablename)) - if(logact&PgLOG.ENDLCK): endtran() elif curtran: curtran += count if curtran > PGDBI['MTRANS']: starttran() - return (ids if ids else count) # @@ -938,8 +896,8 @@ def pgget(tablenames, fields, condition = None, logact = 0): # return a dict reference with keys as field names upon success, values for each field name # are in a list. All lists are the same length with missing values set to None # -def pgmget(tablenames, fields, condition = None, logact = PGDBI['ERRLOG']): - +def pgmget(tablenames, fields, condition = None, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] sqlstr = prepare_select(tablenames, fields, condition, None, logact) ucname = True if logact&PgLOG.UCNAME else False count = pgcnt = 0 @@ -968,10 +926,8 @@ def pgmget(tablenames, fields, condition = None, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgmget: {} record(s) retrieved from {}".format(count, tablenames)) - return records # @@ -982,18 +938,16 @@ def pgmget(tablenames, fields, condition = None, logact = PGDBI['ERRLOG']): # # retrieve one records from tablenames condition dict # -def pghget(tablenames, fields, cnddict, logact = PGDBI['ERRLOG']): - +def pghget(tablenames, fields, cnddict, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] if not tablenames: return PgLOG.pglog("Miss Table name to query", logact) if not fields: return PgLOG.pglog("Nothing to query " + tablenames, logact) if not cnddict: return PgLOG.pglog("Miss condition dict values to query " + tablenames, logact) sqlstr = prepare_select(tablenames, fields, None, list(cnddict), logact) if fields and not re.search(r'limit 1$', sqlstr, re.I): sqlstr += " LIMIT 1" - ucname = True if logact&PgLOG.UCNAME else False - + ucname = True if logact&PgLOG.UCNAME else False values = tuple(cnddict.values()) if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "Query from {} for {}".format(tablenames, values)) - pgcnt = 0 record = {} while True: @@ -1016,7 +970,6 @@ def pghget(tablenames, fields, cnddict, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if record and tablenames and not fields: if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pghget: {} record(s) found from {}".format(record['cntrec'], tablenames)) @@ -1024,7 +977,6 @@ def pghget(tablenames, fields, cnddict, logact = PGDBI['ERRLOG']): elif PgLOG.PGLOG['DBGLEVEL']: cnt = 1 if record else 0 PgLOG.pgdbg(1000, "pghget: {} record retrieved from {}".format(cnt, tablenames)) - return record # @@ -1035,22 +987,19 @@ def pghget(tablenames, fields, cnddict, logact = PGDBI['ERRLOG']): # # retrieve multiple records from tablenames for condition dict # -def pgmhget(tablenames, fields, cnddicts, logact = PGDBI['ERRLOG']): - +def pgmhget(tablenames, fields, cnddicts, logact = None): + if logact is None: logact = PGDBI['ERRLOG'] if not tablenames: return PgLOG.pglog("Miss Table name to query", logact) if not fields: return PgLOG.pglog("Nothing to query " + tablenames, logact) if not cnddicts: return PgLOG.pglog("Miss condition dict values to query " + tablenames, logact) sqlstr = prepare_select(tablenames, fields, None, list(cnddicts), logact) - ucname = True if logact&PgLOG.UCNAME else False - + ucname = True if logact&PgLOG.UCNAME else False v = cnddicts.values() values = list(zip(*v)) cndcnt = len(values) - if PgLOG.PGLOG['DBGLEVEL']: for row in values: PgLOG.pgdbg(1000, "Query from {} for {}".format(tablenames, row)) - colcnt = ccnt = count = pgcnt = 0 cols = [] chrs = [] @@ -1087,11 +1036,9 @@ def pgmhget(tablenames, fields, cnddicts, logact = PGDBI['ERRLOG']): break if ccnt >= cndcnt: break pgcnt += 1 - pgcur.close() - + pgcur.close() if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgmhget: {} record(s) retrieved from {}".format(count, tablenames)) - return records # @@ -1124,17 +1071,15 @@ def prepare_update(tablename, fields, condition = None, cndflds = None): # condition: update conditions for where clause) # return number of rows undated upon success # -def pgupdt(tablename, record, condition, logact = PGDBI['ERRLOG']): - +def pgupdt(tablename, record, condition, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not record: PgLOG.pglog("Nothing updates to " + tablename, logact) if not condition or isinstance(condition, int): PgLOG.pglog("Miss condition to update " + tablename, logact) sqlstr = prepare_update(tablename, list(record), condition) if logact&PgLOG.DODFLT: prepare_default(tablename, record, logact) - values = tuple(record.values()) if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "Update {} for {}".format(tablename, values)) - ucnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -1147,15 +1092,13 @@ def pgupdt(tablename, record, condition, logact = PGDBI['ERRLOG']): if not check_dberror(pgerr, pgcnt, sqlstr, values, logact): return PgLOG.FAILURE else: break - pgcnt += 1 - + pgcnt += 1 if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgupdt: {} record(s) updated to {}".format(ucnt, tablename)) if(logact&PgLOG.ENDLCK): endtran() elif curtran: curtran += ucnt if curtran > PGDBI['MTRANS']: starttran() - return ucnt # @@ -1165,18 +1108,15 @@ def pgupdt(tablename, record, condition, logact = PGDBI['ERRLOG']): # cnddict: condition dict with field names : values # return number of records updated upon success # -def pghupdt(tablename, record, cnddict, logact = PGDBI['ERRLOG']): - +def pghupdt(tablename, record, cnddict, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not record: PgLOG.pglog("Nothing updates to " + tablename, logact) if not cnddict or isinstance(cnddict, int): PgLOG.pglog("Miss condition to update to " + tablename, logact) if logact&PgLOG.DODFLT: prepare_defaults(tablename, record, logact) sqlstr = prepare_update(tablename, list(record), None, list(cnddict)) - values = tuple(record.values()) + tuple(cnddict.values()) - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "Update {} for {}".format(tablename, values)) - ucnt = count = pgcnt = 0 while True: pgcur = pgcursor() @@ -1191,14 +1131,12 @@ def pghupdt(tablename, record, cnddict, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pghupdt: {}/{} record(s) updated to {}".format(ucnt, tablename)) if(logact&PgLOG.ENDLCK): endtran() elif curtran: curtran += ucnt if curtran > PGDBI['MTRANS']: starttran() - return ucnt # @@ -1208,14 +1146,13 @@ def pghupdt(tablename, record, cnddict, logact = PGDBI['ERRLOG']): # cnddicts: condition dict with field names : value lists # return number of records updated upon success # -def pgmupdt(tablename, records, cnddicts, logact = PGDBI['ERRLOG']): - +def pgmupdt(tablename, records, cnddicts, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not records: PgLOG.pglog("Nothing updates to " + tablename, logact) if not cnddicts or isinstance(cnddicts, int): PgLOG.pglog("Miss condition to update to " + tablename, logact) if logact&PgLOG.DODFLT: prepare_defaults(tablename, records, logact) sqlstr = prepare_update(tablename, list(records), None, list(cnddicts)) - fldvals = tuple(records.values()) cntrow = len(fldvals[0]) cndvals = tuple(cnddicts.values()) @@ -1223,10 +1160,8 @@ def pgmupdt(tablename, records, cnddicts, logact = PGDBI['ERRLOG']): if cntcnd != cntrow: return PgLOG.pglog("Field/Condition value counts Miss match {}/{} to update {}".format(cntrow, cntcnd, tablename), logact) v = fldvals + cndvals values = list(zip(*v)) - if PgLOG.PGLOG['DBGLEVEL']: for row in values: PgLOG.pgdbg(1000, "Update {} for {}".format(tablename, row)) - ucnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -1239,16 +1174,13 @@ def pgmupdt(tablename, records, cnddicts, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - pgcur.close() - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgmupdt: {} record(s) updated to {}".format(ucnt, tablename)) if(logact&PgLOG.ENDLCK): endtran() elif curtran: curtran += ucnt if curtran > PGDBI['MTRANS']: starttran() - return ucnt # @@ -1274,12 +1206,11 @@ def prepare_delete(tablename, condition = None, cndflds = None): # condition: delete conditions for where clause # return number of records deleted upon success # -def pgdel(tablename, condition, logact = PGDBI['ERRLOG']): - +def pgdel(tablename, condition, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not condition or isinstance(condition, int): PgLOG.pglog("Miss condition to delete from " + tablename, logact) sqlstr = prepare_delete(tablename, condition) - dcnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -1293,14 +1224,12 @@ def pgdel(tablename, condition, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgdel: {} record(s) deleted from {}".format(dcnt, tablename)) if logact&PgLOG.ENDLCK: endtran() elif curtran: curtran += dcnt if curtran > PGDBI['MTRANS']: starttran() - return dcnt # @@ -1309,15 +1238,13 @@ def pgdel(tablename, condition, logact = PGDBI['ERRLOG']): # cndict: delete condition dict for names : values # return number of records deleted upon success # -def pghdel(tablename, cnddict, logact = PGDBI['ERRLOG']): - +def pghdel(tablename, cnddict, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not cnddict or isinstance(cnddict, int): PgLOG.pglog("Miss condition dict to delete from " + tablename, logact) sqlstr = prepare_delete(tablename, None, list(cnddict)) - values = tuple(cnddict.values()) if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "Delete from {} for {}".format(tablename, values)) - dcnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -1331,14 +1258,12 @@ def pghdel(tablename, cnddict, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pghdel: {} record(s) deleted from {}".format(dcnt, tablename)) if logact&PgLOG.ENDLCK: endtran() elif curtran: curtran += dcnt if curtran > PGDBI['MTRANS']: starttran() - return dcnt # @@ -1347,18 +1272,16 @@ def pghdel(tablename, cnddict, logact = PGDBI['ERRLOG']): # cndicts: delete condition dict for names : value lists # return number of records deleted upon success # -def pgmdel(tablename, cnddicts, logact = PGDBI['ERRLOG']): - +def pgmdel(tablename, cnddicts, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if not cnddicts or isinstance(cnddicts, int): PgLOG.pglog("Miss condition dict to delete from " + tablename, logact) sqlstr = prepare_delete(tablename, None, list(cnddicts)) - v = cnddicts.values() values = list(zip(*v)) if PgLOG.PGLOG['DBGLEVEL']: for row in values: PgLOG.pgdbg(1000, "Delete from {} for {}".format(tablename, row)) - dcnt = pgcnt = 0 while True: pgcur = pgcursor() @@ -1371,27 +1294,23 @@ def pgmdel(tablename, cnddicts, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - pgcur.close() - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgmdel: {} record(s) deleted from {}".format(dcnt, tablename)) if logact&PgLOG.ENDLCK: endtran() elif curtran: curtran += dcnt if curtran > PGDBI['MTRANS']: starttran() - return dcnt # # sqlstr: a complete sql string # return number of record affected upon success # -def pgexec(sqlstr, logact = PGDBI['ERRLOG']): - +def pgexec(sqlstr, logact = None): global curtran + if logact is None: logact = PGDBI['ERRLOG'] if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(100, sqlstr) - ret = pgcnt = 0 while True: pgcur = pgcursor() @@ -1405,14 +1324,12 @@ def pgexec(sqlstr, logact = PGDBI['ERRLOG']): else: break pgcnt += 1 - if PgLOG.PGLOG['DBGLEVEL']: PgLOG.pgdbg(1000, "pgexec: {} record(s) affected for {}".format(ret, sqlstr)) if logact&PgLOG.ENDLCK: endtran() elif curtran: curtran += ret if curtran > PGDBI['MTRANS']: starttran() - return ret # @@ -1460,137 +1377,93 @@ def pgcheck(tablename, logact = 0): # return user.uid upon success, 0 otherwise # def check_user_uid(userno, date = None): - if not userno: return 0 if type(userno) is str: userno = int(userno) - if date is None: datecond = "until_date IS NULL" date = 'today' else: datecond = "(start_date IS NULL OR start_date <= '{}') AND (until_date IS NULL OR until_date >= '{}')".format(date, date) - pgrec = pgget("dssdb.user", "uid", "userno = {} AND {}".format(userno, datecond), PGDBI['ERRLOG']) - if pgrec: return pgrec['uid'] - + if pgrec: return pgrec['uid'] if userno not in NMISSES: PgLOG.pglog("{}: Scientist ID NOT on file for {}".format(userno, date), PgLOG.LGWNEM) NMISSES.append(userno) - # check again if a user is on file with different date range pgrec = pgget("dssdb.user", "uid", "userno = {}".format(userno), PGDBI['ERRLOG']) if pgrec: return pgrec['uid'] - - pgrec = ucar_user_info(userno) - if not pgrec: pgrec = {'userno' : userno, 'stat_flag' : 'M'} - uid = pgadd("dssdb.user", pgrec, (PGDBI['EXITLG']|PgLOG.AUTOID)) + uid = add_missed_user(userno) if uid: PgLOG.pglog("{}: Scientist ID Added as user.uid = {}".format(userno, uid), PgLOG.LGWNEM) - return uid # # return user.uid upon success, 0 otherwise # def get_user_uid(logname, date = None): - if not logname: return 0 if not date: date = 'today' datecond = "until_date IS NULL" else: datecond = "(start_date IS NULL OR start_date <= '{}') AND (until_date IS NULL OR until_date >= '{}')".format(date, date) - pgrec = pgget("dssdb.user", "uid", "logname = '{}' AND {}".format(logname, datecond), PGDBI['ERRLOG']) - if pgrec: return pgrec['uid'] - + if pgrec: return pgrec['uid'] if logname not in LMISSES: PgLOG.pglog("{}: UCAR Login Name NOT on file for {}".format(logname, date), PgLOG.LGWNEM) LMISSES.append(logname) - # check again if a user is on file with different date range pgrec = pgget("dssdb.user", "uid", "logname = '{}'".format(logname), PGDBI['ERRLOG']) if pgrec: return pgrec['uid'] - - pgrec = ucar_user_info(0, logname) - if not pgrec: pgrec = {'logname' : logname, 'stat_flag' : 'M'} - uid = pgadd("dssdb.user", pgrec, (PGDBI['EXITLG']|PgLOG.AUTOID)) + uid = add_missed_user(0, logname) if uid: PgLOG.pglog("{}: UCAR Login Name Added as user.uid = {}".format(logname, uid), PgLOG.LGWNEM) - return uid # -# get ucar user info for given userno (scientist number) or logname (Ucar login) +# get GDEX specialist info from dssdb.dssgrp for given userno or logname # -def ucar_user_info(userno, logname = None): - - MATCH = { - 'upid' : "upid", - 'uid' : "userno", - 'username' : "logname", - 'lastName' : "lstname", - 'firstName' : "fstname", - 'active' : "stat_flag", - 'internalOrg' : "division", - 'externalOrg' : "org_name", - 'country' : "country", - 'forwardEmail' : "email", - 'email' : "ucaremail", - 'phone' : "phoneno" - } - - buf = PgLOG.pgsystem("pgperson " + ("-uid {}".format(userno) if userno else "-username {}".format(logname)), PgLOG.LOGWRN, 20) - if not buf: return None - - pgrec = {} - for line in buf.split('\n'): - ms = re.match(r'^(.+)<=>(.*)$', line) - if ms: - (key, val) = ms.groups() - if key in MATCH: - if key == 'upid' and 'upid' in pgrec: break # get one record only - pgrec[MATCH[key]] = val - +def dssgrp_user_info(userno, logname = None): + cond = "userno = {}".format(userno) if userno else "logname = '{}'".format(logname) + pgrec = pgget("dssgrp", "logname, userno, lstname, fstname, midinit, phoneno", cond, PGDBI['ERRLOG']) if not pgrec: return None + if not pgrec['userno']: del pgrec['userno'] # leave NULL; 0 is never looked up + email = pgrec['logname'] + '@ucar.edu' + pgrec['org_type'] = 'DSS' + pgrec['org_name'] = 'NCAR' + pgrec['country'] = 'UNITED.STATES' + pgrec['email'] = pgrec['ucaremail'] = email + pgrec['stat_flag'] = 'A' + return pgrec - if userno: - pgrec['userno'] = userno - elif pgrec['userno']: - pgrec['userno'] = userno = int(pgrec['userno']) - if pgrec['upid']: pgrec['upid'] = int(pgrec['upid']) - if pgrec['stat_flag']: pgrec['stat_flag'] = 'A' if pgrec['stat_flag'] == "True" else 'C' - if pgrec['email'] and re.search(r'(@|\.)ucar\.edu$', pgrec['email'], re.I): - pgrec['email'] = pgrec['ucaremail'] - pgrec['org_name'] = 'NCAR' - country = pgrec['country'] if 'country' in pgrec else None - pgrec['country'] = set_country_code(pgrec['email'], country) - if pgrec['division']: - val = "NCAR" - else: - val = None - pgrec['org_type'] = get_org_type(val, pgrec['email']) - - buf = PgLOG.pgsystem("pgusername {}".format(pgrec['logname']), PgLOG.LOGWRN, 20) - if not buf: return pgrec - - for line in buf.split('\n'): - ms = re.match(r'^(.+)<=>(.*)$', line) - if ms: - (key, val) = ms.groups() - if key == 'startDate': - m = re.match(r'^(\d+-\d+-\d+)\s', val) - if m: - pgrec['start_date'] = m.group(1) - else: - pgrec['start_date'] = val - - if key == 'endDate': - m = re.match(r'^(\d+-\d+-\d+)\s', val) - if m: - pgrec['until_date'] = m.group(1) - else: - pgrec['until_date'] = val +# +# add a dssdb.user record for a userno/logname missing from the table; +# return the new user.uid +# +def add_missed_user(userno, logname = None): + pgrec = dssgrp_user_info(userno, logname) + if not pgrec: + pgrec = {'stat_flag': 'M', 'lstname': "UNKNOWN", 'fstname': "UNKNOWN"} + if userno: pgrec['userno'] = userno + if logname: + pgrec['logname'] = logname + pgrec['email'] = pgrec['ucaremail'] = logname + '@ucar.edu' + pgrec['org_type'] = 'NCAR' # a UCAR login name outside of the DECS group + pgrec['org_name'] = 'NCAR' + pgrec['country'] = 'UNITED.STATES' + else: + pgrec['org_type'] = "OTHER" + pgrec['org_name'] = pgrec['country'] = "UNKNOWN" + incomplete_user_warning(userno, logname) + return pgadd("dssdb.user", pgrec, (PGDBI['EXITLG']|PgLOG.AUTOID)) - return pgrec +# +# give a one-time reminder that an incomplete dssdb.user record was added +# +def incomplete_user_warning(userno, logname = None): + global USRWARN + if USRWARN: return + USRWARN = 1 + who = logname if logname else userno + PgLOG.pglog("{}: Not in dssdb.dssgrp; incomplete record added to dssdb.user".format(who), PgLOG.LGWNEM) # # set country code for given coutry name or email address @@ -1840,13 +1713,11 @@ def fieldname_string(fnames, dnames = None, anames = None, wflds = None): # go through group tree upward to find a none-empty path, return it or null # def get_group_field_path(gindex, dsid, field): - if gindex: - pgrec = pgget("dsgroup", "pindex, {}".format(field), - "dsid = '{}' AND gindex = {}".format(dsid, gindex), PGDBI['EXITLG']) + pgrec = pgget("dsgroup", f"pindex, {field}", + f"dsid = '{dsid}' AND gindex = {gindex}", PGDBI['EXITLG']) else: - pgrec = pgget("dataset", field, - "dsid = '{}'".format(dsid), PGDBI['EXITLG']) + pgrec = pgget("dataset", field, f"dsid = '{dsid}'", PGDBI['EXITLG']) if pgrec: if pgrec[field]: return pgrec[field] @@ -1858,9 +1729,9 @@ def get_group_field_path(gindex, dsid, field): # # get the specialist info for a given dataset # -def get_specialist(dsid, logact = PGDBI['ERRLOG']): - - if dsid in SPECIALIST: return SPECIALIST['dsid'] +def get_specialist(dsid, logact=None): + if logact is None: logact = PGDBI['ERRLOG'] + if dsid in SPECIALIST: return SPECIALIST[dsid] pgrec = pgget("dsowner, dssgrp", "specialist, lstname, fstname", "specialist = logname AND dsid = '{}' AND priority = 1".format(dsid), logact) @@ -1872,8 +1743,7 @@ def get_specialist(dsid, logact = PGDBI['ERRLOG']): pgrec['specialist'] = "datahelp" pgrec['lstname'] = "Help" pgrec['fstname'] = "Data" - - SPECIALIST['dsid'] = pgrec # cache specialist info for dsowner of dsid + SPECIALIST[dsid] = pgrec # cache specialist info for dsowner of dsid return pgrec # @@ -2097,10 +1967,8 @@ def match_down_path(path, dpaths): # validate is login user is in DECS group # check all node if skpdsg is false, otherwise check non-DSG nodes def validate_decs_group(cmdname, logname, skpdsg): - if skpdsg and PgLOG.PGLOG['DSGHOSTS'] and re.search(r'(^|:){}'.format(PgLOG.PGLOG['HOSTNAME']), PgLOG.PGLOG['DSGHOSTS']): return - if not logname: lgname = PgLOG.PGLOG['CURUID'] - + if not logname: logname = PgLOG.PGLOG['CURUID'] if not pgget("dssgrp", '', "logname = '{}'".format(logname), PgLOG.LGEREX): PgLOG.pglog("{}: Must be in DECS Group to run '{}' on {}".format(logname, cmdname, PgLOG.PGLOG['HOSTNAME']), PgLOG.LGEREX) @@ -2224,13 +2092,11 @@ def add_yearly_wusage(year, records, isarray = 0): # # double quote a array of single or sign delimited strings # -def pgnames(ary, sign = None, joinstr = None): - +def pgnames(ary, sign=None, joinstr=None): pgary = [] for a in ary: pgary.append(pgname(a, sign)) - - if joinstr == None: + if joinstr is None: return pgary else: return joinstr.join(pgary) @@ -2258,11 +2124,11 @@ def pgname(str, sign = None): # get a postgres password for given host, port, dbname, usname # def get_pgpass_password(): - if PGDBI['PWNAME']: return PGDBI['PWNAME'] - pwname = get_baopassword() - if not pwname: pwname = get_pgpassword() - + pwname = get_pgpassword() + if not pwname: pwname = get_baopassword() + if not pwname: + PgLOG.pglog("Unable to find password for {} in .pgpass or OpenBao".format(PGDBI['DBNAME']), PGDBI['ERRLOG']) return pwname def get_pgpassword(): @@ -2283,7 +2149,6 @@ def get_baopassword(): # Reads the .pgpass file and returns a dictionary of credentials. # def read_pgpass(): - pgpass = PgLOG.PGLOG['DSSHOME'] + '/.pgpass' if not op.isfile(pgpass): pgpass = PgLOG.PGLOG['GDEXHOME'] + '/.pgpass' try: @@ -2293,21 +2158,20 @@ def read_pgpass(): if not line or line.startswith("#"): continue dbhost, dbport, dbname, lnname, pwname = line.split(":") DBPASS[(dbhost, dbport, dbname, lnname)] = pwname - except Exception as e: - PgLOG.pglog(str(e), PGDBI['ERRLOG']) + except Exception: + pass # # Reads OpenBao secrets and returns a dictionary of credentials. # def read_openbao(): - dbname = PGDBI['DBNAME'] DBBAOS[dbname] = {} url = 'https://bao.k8s.ucar.edu/' baopath = { - 'ivaddb' : 'gdex/pgdb03', - 'ispddb' : 'gdex/pgdb03', - 'default' : 'gdex/pgdb01' + 'ivaddb': 'gdex/pgdb03', + 'ispddb': 'gdex/pgdb03', + 'default': 'gdex/pgdb01' } dbpath = baopath[dbname] if dbname in baopath else baopath['default'] client = hvac.Client(url=PGDBI.get('BAOURL')) @@ -2318,9 +2182,8 @@ def read_openbao(): mount_point='kv', raise_on_deleted_version=False ) - except Exception as e: - return PgLOG.pglog(str(e), PGDBI['ERRLOG']) - + except Exception: + return baos = read_response['data']['data'] for key in baos: ms = re.match(r'^(\w*)pass(\w*)$', key) diff --git a/src/rda_python_common/PgFile.py b/src/rda_python_common/PgFile.py index 925b26b..dc16e21 100644 --- a/src/rda_python_common/PgFile.py +++ b/src/rda_python_common/PgFile.py @@ -23,6 +23,7 @@ import time import glob import json +import hashlib from . import PgLOG from . import PgUtil from . import PgSIG @@ -59,9 +60,7 @@ TARSTR = '|'.join(PGTARS) DELDIRS = {} -TASKIDS = {} # cache unfinished -MD5CMD = 'md5sum' -SHA512CMD = 'sha512sum' +TASKIDS = {} # cache unfinished LHOST = "localhost" OHOST = PgLOG.PGLOG['OBJCTSTR'] BHOST = PgLOG.PGLOG['BACKUPNM'] @@ -69,8 +68,6 @@ OBJCTCMD = "isd_s3_cli" BACKCMD = "dsglobus" -HLIMIT = 0 # HTAR file count limit -BLIMIT = 2 # minimum back tar file size in DB DIRLVLS = 0 # record how many errors happen for working with HPSS, local or remote machines @@ -163,11 +160,11 @@ def errlog(msg, etype, retry = 0, logact = 0): # # Return 1 if successful 0 if failed with error message generated in PgLOG.pgsystem() cached in PgLOG.PGLOG['SYSERR'] # -def copy_gdex_file(tofile, fromfile, tohost = LHOST, fromhost = LHOST, logact = 0): - +def copy_gdex_file(tofile, fromfile, tohost = None, fromhost = None, logact = 0): + if tohost is None: tohost = LHOST + if fromhost is None: fromhost = LHOST thost = strip_host_name(tohost) fhost = strip_host_name(fromhost) - if PgUtil.pgcmp(thost, fhost, 1) == 0: if PgUtil.pgcmp(thost, LHOST, 1) == 0: return local_copy_local(tofile, fromfile, logact) @@ -175,9 +172,9 @@ def copy_gdex_file(tofile, fromfile, tohost = LHOST, fromhost = LHOST, logact = if PgUtil.pgcmp(thost, OHOST, 1) == 0: return local_copy_object(tofile, fromfile, None, None, logact) elif PgUtil.pgcmp(thost, BHOST, 1) == 0: - return local_copy_backup(tofile, fromfile, QPOINTS['B'], logact) + return wait_copy_backup(tofile, fromfile, QPOINTS['B'], logact) elif PgUtil.pgcmp(thost, DHOST, 1) == 0: - return local_copy_backup(tofile, fromfile, QPOINTS['D'], logact) + return wait_copy_backup(tofile, fromfile, QPOINTS['D'], logact) else: return local_copy_remote(tofile, fromfile, tohost, logact) elif PgUtil.pgcmp(thost, LHOST, 1) == 0: @@ -189,8 +186,7 @@ def copy_gdex_file(tofile, fromfile, tohost = LHOST, fromhost = LHOST, logact = return backup_copy_local(tofile, fromfile, QPOINTS['D'], logact) else: return remote_copy_local(tofile, fromfile, fromhost) - - return errlog("{}-{}->{}-{}: Cannot copy file".format(fhost, fromfile, thost, tofile), 'O', 1, PgLOG.LGEREX) + return errlog("{}-{}->{}-{}: Cannot copy file".format(fhost, fromfile, thost, tofile), 'O', 1, PgLOG.LGEREX) copy_rda_file = copy_gdex_file @@ -298,24 +294,21 @@ def local_copy_remote(tofile, fromfile, host, logact = 0): # meta - reference to metadata hash # def local_copy_object(tofile, fromfile, bucket = None, meta = None, logact = 0): - if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] if meta is None: meta = {} if 'user' not in meta: meta['user'] = PgLOG.PGLOG['CURUID'] if 'group' not in meta: meta['group'] = PgLOG.PGLOG['GDEXGRP'] uinfo = json.dumps(meta) - - finfo = check_local_file(fromfile, 0, logact) + finfo = check_local_file(fromfile, 0, logact|PgLOG.PFSIZE) if not finfo: if finfo != None: return PgLOG.FAILURE return lmsg(fromfile, "{} to copy to {}-{}".format(PgLOG.PGLOG['MISSFILE'], OHOST, tofile), logact) - if not logact&PgLOG.OVRIDE: tinfo = check_object_file(tofile, bucket, 0, logact) if tinfo and tinfo['data_size'] > 0: return PgLOG.pglog("{}-{}-{}: file exists already".format(OHOST, bucket, tofile), logact) - - cmd = "{} ul -lf {} -b {} -k {} -md '{}'".format(OBJCTCMD, fromfile, bucket, tofile, uinfo) + ocmd = OBJCTCMD + cmd = "{} ul -lf {} -b {} -k {} -md '{}'".format(ocmd, fromfile, bucket, tofile, uinfo) for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDBTH) tinfo = check_object_file(tofile, bucket, 0, logact) @@ -324,9 +317,7 @@ def local_copy_object(tofile, fromfile, bucket = None, meta = None, logact = 0): return PgLOG.SUCCESS elif tinfo != None: break - errlog("Error Execute: {}\n{}".format(cmd, buf), 'O', loop, logact) - return PgLOG.FAILURE # @@ -337,7 +328,7 @@ def local_copy_object(tofile, fromfile, bucket = None, meta = None, logact = 0): # topoint - target endpoint name, 'gdex-glade', 'gdex-quasar' or 'gdex-quasar-dgdexta' # frompoint - source endpoint name, the same choices as the topoint # -def quasar_multiple_trasnfer(tofiles, fromfiles, topoint, frompoint, logact = 0): +def quasar_multiple_transfer(tofiles, fromfiles, topoint, frompoint, logact = 0): ret = PgLOG.FAILURE @@ -356,7 +347,8 @@ def quasar_multiple_trasnfer(tofiles, fromfiles, topoint, frompoint, logact = 0) label = f"{ENDPOINTS[frompoint]} to {ENDPOINTS[topoint]} {action}" verify_checksum = True - cmd = f'{BACKCMD} {action} -se {source_endpoint} -de {destination_endpoint} --label "{label}"' + bcmd = BACKCMD + cmd = f'{bcmd} {action} -se {source_endpoint} -de {destination_endpoint} --label "{label}"' if verify_checksum: cmd += ' -vc' cmd += ' --batch -' @@ -370,38 +362,37 @@ def quasar_multiple_trasnfer(tofiles, fromfiles, topoint, frompoint, logact = 0) return ret +# backward compatible alias for the previously misspelled function name +quasar_multiple_trasnfer = quasar_multiple_transfer + # # Copy a file from a Globus endpoint to another -# tofile - target file name, leading with /dsnnn.n/ on Quasar and +# tofile - target file name, leading with /dsnnn.n/ on Quasar and # leading with /data/ or /decsdata/ on local glade disk # fromfile - source file, the same format as the tofile # topoint - target endpoint name, 'gdex-glade', 'gdex-quasar' or 'gdex-quasar-dgdexta' # frompoint - source endpoint name, the same choices as the topoint # def endpoint_copy_endpoint(tofile, fromfile, topoint, frompoint, logact = 0): - ret = PgLOG.FAILURE finfo = check_globus_file(fromfile, frompoint, 0, logact) if not finfo: if finfo != None: return ret return lmsg(fromfile, "{} to copy {} file to {}-{}".format(PgLOG.PGLOG['MISSFILE'], frompoint, topoint, tofile), logact) - if not logact&PgLOG.OVRIDE: tinfo = check_globus_file(tofile, topoint, 0, logact) if tinfo and tinfo['data_size'] > 0: return PgLOG.pglog("{}-{}: file exists already".format(topoint, tofile), logact) - action = 'transfer' - cmd = f'{BACKCMD} {action} -se {frompoint} -de {topoint} -sf {fromfile} -df {tofile} -vc' - + bcmd = BACKCMD + cmd = f'{bcmd} {action} -se {frompoint} -de {topoint} -sf {fromfile} -df {tofile} -vc' task = submit_globus_task(cmd, topoint, logact) if task['stat'] == 'S': ret = PgLOG.SUCCESS elif task['stat'] == 'A': TASKIDS["{}-{}".format(topoint, tofile)] = task['id'] ret = PgLOG.FINISH - return ret # @@ -445,15 +436,13 @@ def submit_globus_task(cmd, endpoint, logact = 0, qstr = None): # if PgLOG.NOWAIT presents and Details is neither OK nor Queued # def check_globus_status(taskid, endpoint = None, logact = 0): - ret = 'U' if not taskid: return ret if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] mp = r'Status:\s+({})'.format('|'.join(QSTATS.values())) - - cmd = f"{BACKCMD} get-task {taskid}" + bcmd = BACKCMD + cmd = f"{bcmd} get-task {taskid}" astats = ['OK', 'Queued'] - for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDRET) if buf: @@ -468,20 +457,18 @@ def check_globus_status(taskid, endpoint = None, logact = 0): if logact&PgLOG.NOWAIT: errmsg = "{}: Cancel Task due to {}:\n{}".format(taskid, detail, buf) errlog(errmsg, 'B', 1, logact) - ccmd = f"{BACKCMD} cancel-task {taskid}" + ccmd = f"{bcmd} cancel-task {taskid}" PgLOG.pgsystem(ccmd, logact, 7) else: time.sleep(PgSIG.PGSIG['ETIME']) continue break - errmsg = "Error Execute: " + cmd if PgLOG.PGLOG['SYSERR']: errmsg = "\n" + PgLOG.PGLOG['SYSERR'] (hstat, msg) = host_down_status('', QHOSTS[endpoint], 1, logact) if hstat: errmsg += "\n" + msg errlog(errmsg, 'B', loop, logact) - if ret == 'S' or ret == 'A': ECNTS['B'] = 0 # reset error count return ret @@ -536,6 +523,17 @@ def local_copy_backup(tofile, fromfile, endpoint = None, logact = 0): if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] return endpoint_copy_endpoint(tofile, fromfile, endpoint, 'gdex-glade', logact) +# +# Copy a local file to Quasar backup tape system and wait until the Globus task is done +# for callers that cannot handle a PgLOG.FINISH return +# +def wait_copy_backup(tofile, fromfile, endpoint = None, logact = 0): + + if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] + ret = local_copy_backup(tofile, fromfile, endpoint, logact) + if ret == PgLOG.FINISH: ret = check_globus_finished(tofile, endpoint, logact) + return ret + # # Copy a Quasar backup file to local Globus endpoint # @@ -544,9 +542,10 @@ def local_copy_backup(tofile, fromfile, endpoint = None, logact = 0): # endpoint - endpoint name on Quasar Backup Server # def backup_copy_local(tofile, fromfile, endpoint = None, logact = 0): - if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] - return endpoint_copy_endpoint(tofile, fromfile, 'gdex-glade', endpoint, logact) + ret = endpoint_copy_endpoint(tofile, fromfile, 'gdex-glade', endpoint, logact) + if ret == PgLOG.FINISH: ret = check_globus_finished(tofile, 'gdex-glade', logact) + return ret # # Copy a remote file to local @@ -556,23 +555,19 @@ def backup_copy_local(tofile, fromfile, endpoint = None, logact = 0): # host - remote host name # def remote_copy_local(tofile, fromfile, host, logact = 0): - cmd = PgLOG.get_sync_command(host) finfo = check_remote_file(fromfile, host, 0, logact) + target = tofile if not finfo: if finfo != None: return PgLOG.FAILURE return errlog("{}-{}: {} to copy to {}".format(host, fromfile, PgLOG.PGLOG['MISSFILE'], tofile), 'R', 1, logact) - - target = tofile ms = re.match(r'^(.+)/$', tofile) if ms: dir = ms.group(1) tofile += op.basename(fromfile) else: dir = get_local_dirname(tofile) - if not make_local_directory(dir, logact): return PgLOG.FAILURE - cmd += " -g {} {}".format(fromfile, target) loop = reset = 0 while (loop-reset) < 2: @@ -587,11 +582,9 @@ def remote_copy_local(tofile, fromfile, host, logact = 0): return PgLOG.SUCCESS elif info != None: break - errlog(PgLOG.PGLOG['SYSERR'], 'L', (loop - reset), logact) if loop == 0: reset = reset_local_info(tofile, info, logact) loop += 1 - return PgLOG.FAILURE # @@ -602,15 +595,14 @@ def remote_copy_local(tofile, fromfile, host, logact = 0): # bucket - bucket name on Object store # def object_copy_local(tofile, fromfile, bucket = None, logact = 0): - ret = PgLOG.FAILURE if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] finfo = check_object_file(fromfile, bucket, 0, logact) if not finfo: if finfo != None: return ret return lmsg(fromfile, "{}-{} to copy to {}".format(OHOST, PgLOG.PGLOG['MISSFILE'], tofile), logact) - - cmd = "{} go -k {} -b {}".format(OBJCTCMD, fromfile, bucket) + ocmd = OBJCTCMD + cmd = "{} go -k {} -b {}".format(ocmd, fromfile, bucket) fromname = op.basename(fromfile) toname = op.basename(tofile) if toname == tofile: @@ -621,24 +613,20 @@ def object_copy_local(tofile, fromfile, bucket = None, logact = 0): loop = reset = 0 while (loop-reset) < 2: buf = PgLOG.pgsystem(cmd, logact, CMDBTH) - info = check_local_file(fromname, 143, logact) # 1+2+4+8+128 + info = check_local_file(fromname, 143, logact|PgLOG.PFSIZE) # 1+2+4+8+128 if info: if info['data_size'] == finfo['data_size']: set_local_mode(fromfile, info['isfile'], 0, info['mode'], info['logname'], logact) if toname == fromname or move_local_file(toname, fromname, logact): ret = PgLOG.SUCCESS break - - elif info != None: break - errlog("Error Execute: {}\n{}".format(cmd, buf), 'L', (loop - reset), logact) if loop == 0: reset = reset_local_info(tofile, info, logact) loop += 1 if odir and odir != dir: change_local_directory(odir, logact) - return ret # @@ -750,41 +738,37 @@ def delete_remote_file(file, host, logact = 0): # Delete a file on object store # def delete_object_file(file, bucket = None, logact = 0): - if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] + ocmd = OBJCTCMD for loop in range(2): list = object_glob(file, bucket, 0, logact) if not list: return PgLOG.FAILURE errmsg = None for key in list: - cmd = "{} dl {} -b {}".format(OBJCTCMD, key, bucket) + cmd = "{} dl {} -b {}".format(ocmd, key, bucket) if not PgLOG.pgsystem(cmd, logact, CMDERR): errmsg = PgLOG.PGLOG['SYSERR'] break - list = object_glob(file, bucket, 0, logact) if not list: return PgLOG.SUCCESS if errmsg: errlog(errmsg, 'O', loop, logact) - return PgLOG.FAILURE # # Delete a backup file on Quasar Server # def delete_backup_file(file, endpoint = None, logact = 0): - if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] info = check_backup_file(file, endpoint, 0, logact) if not info: return PgLOG.FAILURE - - cmd = f"{BACKCMD} delete -ep {endpoint} -tf {file}" + bcmd = BACKCMD + cmd = f"{bcmd} delete -ep {endpoint} -tf {file}" task = submit_globus_task(cmd, endpoint, logact) if task['stat'] == 'S': return PgLOG.SUCCESS elif task['stat'] == 'A': TASKIDS["{}-{}".format(endpoint, file)] = task['id'] return PgLOG.FINISH - return PgLOG.FAILURE # @@ -951,7 +935,6 @@ def move_remote_file(tofile, fromfile, host, logact = 0): # frombucket - original bucket name # def move_object_file(tofile, fromfile, tobucket, frombucket, logact = 0): - ret = PgLOG.FAILURE if not tobucket: tobucket = PgLOG.PGLOG['OBJCTBKT'] if not frombucket: frombucket = tobucket @@ -969,12 +952,11 @@ def move_object_file(tofile, fromfile, tobucket, frombucket, logact = 0): return errlog("{}-{}: Object File exists, cannot move {}-{} to it".format(tobucket, tofile, frombucket, fromfile), 'R', 1, logact) elif tinfo != None: return PgLOG.FAILURE - - cmd = "{} mv -b {} -db {} -k {} -dk {}".format(OBJCTCMD, frombucket, tobucket, fromfile, tofile) - ucmd = "{} gm -k {} -b {}".format(OBJCTCMD, fromfile, frombucket) + ocmd = OBJCTCMD + cmd = "{} mv -b {} -db {} -k {} -dk {}".format(ocmd, frombucket, tobucket, fromfile, tofile) + ucmd = "{} gm -k {} -b {}".format(ocmd, fromfile, frombucket) ubuf = PgLOG.pgsystem(ucmd, PgLOG.LOGWRN, CMDRET) if ubuf and re.match(r'^\{', ubuf): cmd += " -md '{}'".format(ubuf) - for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDBTH) tinfo = check_object_file(tofile, tobucket, 0, logact) @@ -983,9 +965,7 @@ def move_object_file(tofile, fromfile, tobucket, frombucket, logact = 0): return PgLOG.SUCCESS elif tinfo != None: break - errlog("Error Execute: {}\n{}".format(cmd, buf), 'O', loop, logact) - return PgLOG.FAILURE # @@ -997,7 +977,6 @@ def move_object_file(tofile, fromfile, tobucket, frombucket, logact = 0): # frombucket - original bucket name # def move_object_path(topath, frompath, tobucket, frombucket, logact = 0): - ret = PgLOG.FAILURE if not tobucket: tobucket = PgLOG.PGLOG['OBJCTBKT'] if not frombucket: frombucket = tobucket @@ -1010,15 +989,13 @@ def move_object_path(topath, frompath, tobucket, frombucket, logact = 0): return PgLOG.SUCCESS else: return errlog("{}-{}: {} to move".format(frombucket, frompath, PgLOG.PGLOG['MISSFILE']), 'R', 1, logact) - - cmd = "{} mv -b {} -db {} -k {} -dk {}".format(OBJCTCMD, frombucket, tobucket, frompath, topath) - + ocmd = OBJCTCMD + cmd = "{} mv -b {} -db {} -k {} -dk {}".format(ocmd, frombucket, tobucket, frompath, topath) for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDBTH) fcnt = check_object_path(frompath, frombucket, logact) if not fcnt: return PgLOG.SUCCESS errlog("Error Execute: {}\n{}".format(cmd, buf), 'O', loop, logact) - return PgLOG.FAILURE # @@ -1029,7 +1006,6 @@ def move_object_path(topath, frompath, tobucket, frombucket, logact = 0): # endpoint - Globus endpoint # def move_backup_file(tofile, fromfile, endpoint = None, logact = 0): - ret = PgLOG.FAILURE if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] finfo = check_backup_file(fromfile, endpoint, 0, logact) @@ -1041,14 +1017,13 @@ def move_backup_file(tofile, fromfile, endpoint = None, logact = 0): return PgLOG.SUCCESS else: return errlog("{}: {} to move".format(fromfile, PgLOG.PGLOG['MISSFILE']), 'B', 1, logact) - if tinfo: if tinfo['data_size'] > 0 and not logact&PgLOG.OVRIDE: return errlog("{}: File exists, cannot move {} to it".format(tofile, fromfile), 'B', 1, logact) elif tinfo != None: return ret - - cmd = f"{BACKCMD} rename -ep {endpoint} --old-path {fromfile} --new-path {tofile}" + bcmd = BACKCMD + cmd = f"{bcmd} rename -ep {endpoint} --old-path {fromfile} --new-path {tofile}" loop = 0 while loop < 2: buf = PgLOG.pgsystem(cmd, logact, CMDRET) @@ -1065,7 +1040,6 @@ def move_backup_file(tofile, fromfile, endpoint = None, logact = 0): if hstat: errmsg += "\n" + msg errlog(errmsg, 'B', loop, logact) loop += 1 - if ret == PgLOG.SUCCESS: ECNTS['B'] = 0 # reset error count return ret @@ -1167,7 +1141,6 @@ def make_backup_directory(dir, endpoint, logact = 0): # Make a quasar directory recursively # def make_one_backup_directory(dir, odir, endpoint = None, logact = 0): - if not dir or dir == '/': return PgLOG.SUCCESS if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] info = check_backup_file(dir, endpoint, 0, logact) @@ -1176,11 +1149,11 @@ def make_one_backup_directory(dir, odir, endpoint = None, logact = 0): return PgLOG.SUCCESS elif info != None: return PgLOG.FAILURE - if not odir: odir = dir if not make_one_backup_directory(op.dirname(dir), odir, endpoint, logact): return PgLOG.FAILURE - - cmd = f"{BACKCMD} mkdir -ep {endpoint} -p {dir}" + bcmd = BACKCMD + cmd = f"{bcmd} mkdir -ep {endpoint} -p {dir}" + ret = PgLOG.FAILURE for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDRET) syserr = PgLOG.PGLOG['SYSERR'] @@ -1199,7 +1172,6 @@ def make_one_backup_directory(dir, odir, endpoint = None, logact = 0): (hstat, msg) = host_down_status('', QHOSTS[endpoint], 1, logact) if hstat: errmsg += "\n" + msg errlog(errmsg, 'B', loop, logact) - if ret == PgLOG.SUCCESS: ECNTS['B'] = 0 # reset error count return ret @@ -1207,17 +1179,8 @@ def make_one_backup_directory(dir, odir, endpoint = None, logact = 0): # check and return 1 if a root directory # def is_root_directory(dir, etype, host = None, action = None, logact = 0): - ret = cnt = 0 - - if etype == 'H': - ms = re.match(r'^({})(.*)$'.format(PgLOG.PGLOG['ALLROOTS']), dir) - if ms: - m2 = ms.group(2) - if not m2 or m2 == '/': ret = 1 - else: - cnt = 2 - elif re.match(r'^{}'.format(PgLOG.PGLOG['DSSDATA']), dir): + if re.match(r'^{}'.format(PgLOG.PGLOG['DSSDATA']), dir): ms = re.match(r'^({})(.*)$'.format(PgLOG.PGLOG['GPFSROOTS']), dir) if ms: m2 = ms.group(2) @@ -1231,17 +1194,14 @@ def is_root_directory(dir, etype, host = None, action = None, logact = 0): if not m2 or m2 == '/': ret = 1 else: cnt = 2 - if cnt and re.match(r'^(/[^/]+){0,%d}(/*)$' % cnt, dir): ret = 1 - if ret and action: cnt = 0 errmsg = "{}: Cannot {} from {}".format(dir, action, PgLOG.PGLOG['HOSTNAME']) (hstat, msg) = host_down_status(dir, host, 0, logact) if hstat: errmsg += "\n" + msg errlog(errmsg, etype, 1, logact|PgLOG.ERRLOG) - return ret # @@ -1261,23 +1221,31 @@ def set_gdex_mode(file, isfile, host, nmode = None, omode = None, logname = None # set mode for given local directory or file # def set_local_mode(file, isfile = 1, nmode = 0, omode = 0, logname = None, logact = 0): - if not nmode: nmode = (PgLOG.PGLOG['FILEMODE'] if isfile else PgLOG.PGLOG['EXECMODE']) if not (omode and logname): info = check_local_file(file, 6) if not info: - if info != None: return PgLOG.FAILURE - return lmsg(file, "{} to set mode({})".format(PgLOG.PGLOG['MISSFILE'], PgLOG.int2base(nmode, 8)), logact) + if info != None: return PgLOG.FAILURE + return lmsg(file, "{} to set mode({})".format(PgLOG.PGLOG['MISSFILE'], PgLOG.int2base(nmode, 8)), logact) omode = info['mode'] logname = info['logname'] - if nmode == omode: return PgLOG.SUCCESS - + if logact and logact&PgLOG.EXITLG: logact &= ~PgLOG.EXITLG + euid = os.geteuid() + if euid and logname and pwd.getpwnam(logname).pw_uid != euid: + # only the file owner (or root) can chmod; try the pgstart_ setuid wrapper + cmd = "chmod {} {}".format(PgLOG.int2base(nmode, 8), file) + wcmd = PgLOG.get_local_command(cmd, logname) + msg = "{}: Cannot set mode({}) for file owned by {}".format(file, PgLOG.int2base(nmode, 8), logname) + if wcmd != cmd: + if PgLOG.pgsystem(wcmd, logact, 257): return PgLOG.SUCCESS + if PgLOG.PGLOG['SYSERR']: msg += "\n" + PgLOG.PGLOG['SYSERR'] + PgLOG.pglog(msg, PgLOG.LOGWRN) + return PgLOG.SUCCESS try: os.chmod(file, nmode) except Exception as e: return errlog(str(e), 'L', 1, logact) - return PgLOG.SUCCESS # @@ -1565,10 +1533,9 @@ def strip_host_name(host): # # Return a dict of file info, or None if file not exists # -def check_gdex_file(file, host = LHOST, opt = 0, logact = 0): - +def check_gdex_file(file, host = None, opt = 0, logact = 0): + if host is None: host = LHOST shost = strip_host_name(host) - if PgUtil.pgcmp(shost, LHOST, 1) == 0: return check_local_file(file, opt, logact) elif PgUtil.pgcmp(shost, OHOST, 1) == 0: @@ -1776,12 +1743,14 @@ def remote_file_stat(line, opt): # Return a dict of file info, or None if file not exists # def check_object_file(file, bucket = None, opt = 0, logact = 0): - if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] ret = None if not file: return ret - cmd = "{} lo {} -b {}".format(OBJCTCMD, file, bucket) - ucmd = "{} gm -k {} -b {}".format(OBJCTCMD, file, bucket) if opt&14 else None + ms = re.match(r'^(.+)/$', file) + if ms: file = ms.group(1) # remove ending '/' in case + ocmd = OBJCTCMD + cmd = "{} lo {} -b {}".format(ocmd, file, bucket) + ucmd = "{} gm -k {} -b {}".format(ocmd, file, bucket) if opt&14 else None loop = 0 while loop < 2: buf = PgLOG.pgsystem(cmd, PgLOG.LOGWRN, CMDRET) @@ -1789,14 +1758,23 @@ def check_object_file(file, bucket = None, opt = 0, logact = 0): if re.match(r'^\[\]', buf): break if re.match(r'^\[\{', buf): ary = json.loads(buf) - cnt = len(ary) - if cnt > 1: return PgLOG.pglog("{}-{}: {} records returned\n{}".format(bucket, file, cnt, buf), logact|PgLOG.ERRLOG) hash = ary[0] uhash = None if ucmd: ubuf = PgLOG.pgsystem(ucmd, PgLOG.LOGWRN, CMDRET) if ubuf and re.match(r'^\{', ubuf): uhash = json.loads(ubuf) ret = object_file_stat(hash, uhash, opt) + if ret: + cnt = len(ary) + if cnt > 1 or hash['Key'] != file: + ret['count'] = cnt + ret['fname'] = op.basename(file) + ret['isfile'] = 0 + size = 0 + for a in ary: + size += int(a['Size']) + ret['data_size'] = size + uhash = None break if opt&64: return PgLOG.FAILURE errmsg = "Error Execute: {}\n{}".format(cmd, PgLOG.PGLOG['SYSERR']) @@ -1804,7 +1782,6 @@ def check_object_file(file, bucket = None, opt = 0, logact = 0): if hstat: errmsg += "\n" + msg errlog(errmsg, 'O', loop, logact) loop += 1 - if loop > 1: return PgLOG.FAILURE ECNTS['O'] = 0 # reset error count return ret @@ -1817,11 +1794,11 @@ def check_object_file(file, bucket = None, opt = 0, logact = 0): # Return count of object key names, 0 if not file exists; None if error checking # def check_object_path(path, bucket = None, logact = 0): - if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] ret = None if not path: return ret - cmd = "{} lo {} -ls -b {}".format(OBJCTCMD, path, bucket) + ocmd = OBJCTCMD + cmd = "{} lo {} -ls -b {}".format(ocmd, path, bucket) loop = 0 while loop < 2: buf = PgLOG.pgsystem(cmd, PgLOG.LOGWRN, CMDRET) @@ -1833,7 +1810,6 @@ def check_object_path(path, bucket = None, logact = 0): if hstat: errmsg += "\n" + msg errlog(errmsg, 'O', loop, logact) loop += 1 - ECNTS['O'] = 0 # reset error count return ret @@ -1878,13 +1854,13 @@ def object_file_stat(hash, uhash, opt): # Return a dict of file info, or None if file not exists # def check_backup_file(file, endpoint = None, opt = 0, logact = 0): - ret = None if not file: return ret if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] bdir = op.dirname(file) bfile = op.basename(file) - cmd = f"{BACKCMD} ls -ep {endpoint} -p {bdir} --filter {bfile}" + bcmd = BACKCMD + cmd = f"{bcmd} ls -ep {endpoint} -p {bdir} --filter {bfile}" ccnt = loop = 0 while loop < 2: buf = PgLOG.pgsystem(cmd, logact, CMDRET) @@ -1915,7 +1891,6 @@ def check_backup_file(file, endpoint = None, opt = 0, logact = 0): if hstat: errmsg += "\n" + msg errlog(errmsg, 'B', loop, logact) loop += 1 - if ret: ECNTS['B'] = 0 # reset error count return ret @@ -2061,7 +2036,6 @@ def check_ftp_file(file, opt = 0, name = None, pswd = None, logact = 0): # local function to get stat of a file on ftp server # def ftp_file_stat(line, opt): - items = re.split(r'\s+', line) if len(items) < 9: return None ms = re.match(r'^([d\-])([\w\-]{9})$', items[0]) @@ -2074,7 +2048,7 @@ def ftp_file_stat(line, opt): if opt&17: dy = int(items[6]) mn = PgUtil.get_month(items[5]) - if re.match(r'^\d+$', items[7]): + if items[7].isdigit(): yr = int(items[7]) mtime = "00:00:00" else: @@ -2085,16 +2059,13 @@ def ftp_file_stat(line, opt): yr = int(ms.group(1)) cm = int(ms.group(2)) # current month if cm < mn: yr -= 1 # previous year - mdate = "{}-{:02}-{:02}".format(yr, mn, dy) if opt&1: info['date_modified'] = mdate info['time_modified'] = mtime if opt&16: info['week_day'] = PgUtil.get_weekday(mdate) - if opt&2: info['logname'] = items[2] if opt&8: info['group'] = items[3] - return info # @@ -2201,12 +2172,12 @@ def remote_glob(dir, host, opt = 0, logact = 0): # Return: a dict with filenames as keys, or None if not exists # def object_glob(dir, bucket = None, opt = 0, logact = 0): - flist = {} if not bucket: bucket = PgLOG.PGLOG['OBJCTBKT'] ms = re.match(r'^(.+)/$', dir) if ms: dir = ms.group(1) - cmd = "{} lo {} -b {}".format(OBJCTCMD, dir, bucket) + ocmd = OBJCTCMD + cmd = "{} lo {} -b {}".format(ocmd, dir, bucket) ary = err = None buf = PgLOG.pgsystem(cmd, PgLOG.LOGWRN, CMDRET) if buf: @@ -2222,16 +2193,14 @@ def object_glob(dir, bucket = None, opt = 0, logact = 0): return PgLOG.FAILURE else: return flist - for hash in ary: uhash = None if opt&10: - ucmd = "{} gm -l {} -b {}".format(OBJCTCMD, hash['Key'], bucket) + ucmd = "{} gm -l {} -b {}".format(ocmd, hash['Key'], bucket) ubuf = PgLOG.pgsystem(ucmd, PgLOG.LOGWRN, CMDRET) if ubuf and re.match(r'^\{.+', ubuf): uhash = json.loads(ubuf) info = object_file_stat(hash, uhash, opt) if info: flist[hash['Key']] = info - return flist # @@ -2248,11 +2217,10 @@ def object_glob(dir, bucket = None, opt = 0, logact = 0): # Return: a dict with filenames as keys, or None if not exists # def backup_glob(dir, endpoint = None, opt = 0, logact = 0): - if not dir: return None if not endpoint: endpoint = PgLOG.PGLOG['BACKUPEP'] - - cmd = f"{BACKCMD} ls -ep {endpoint} -p {dir}" + bcmd = BACKCMD + cmd = f"{bcmd} ls -ep {endpoint} -p {dir}" flist = {} for loop in range(2): buf = PgLOG.pgsystem(cmd, logact, CMDRET) @@ -2278,7 +2246,6 @@ def backup_glob(dir, endpoint = None, opt = 0, logact = 0): (hstat, msg) = host_down_status('', QHOSTS[endpoint], 0, logact) if hstat: errmsg += "\n" + msg errlog(errmsg, 'B', loop, logact) - if flist: ECNTS['B'] = 0 # reset error count return flist @@ -2313,27 +2280,33 @@ def get_file_mode(perm): # Return: one or a array of 128-bits md5 'fingerprint' None if failed # def get_md5sum(file, count = 0, logact = 0): - - cmd = MD5CMD + ' ' - if count > 0: checksum = [None]*count for i in range(count): if op.isfile(file[i]): - chksm = PgLOG.pgsystem(cmd + file[i], logact, 20) - if chksm: - ms = re.search(r'(\w{32})', chksm) - if ms: checksum[i] = ms.group(1) + checksum[i] = _file_md5(file[i], logact) else: checksum = None if op.isfile(file): - chksm = PgLOG.pgsystem(cmd + file, logact, 20) - if chksm: - ms = re.search(r'(\w{32})', chksm) - if ms: checksum = ms.group(1) - + checksum = _file_md5(file, logact) return checksum +# +# Compute MD5 hex digest of a given file, reading in 1 MiB chunks +# +# Return the hex digest string, or None on read error +# +def _file_md5(path, logact = 0): + try: + h = hashlib.md5() + with open(path, 'rb') as fh: + for chunk in iter(lambda: fh.read(1048576), b''): + h.update(chunk) + return h.hexdigest() + except OSError as e: + PgLOG.pglog("Error md5sum {}: {}".format(path, str(e)), logact) + return None + # # Evaluate md5 checksums and compare them for two given files # @@ -2342,7 +2315,6 @@ def get_md5sum(file, count = 0, logact = 0): # Return: 0 if same and 1 if not # def compare_md5sum(file1, file2, logact = 0): - if op.isdir(file1) or op.isdir(file2): files1 = get_directory_files(file1) fcnt1 = len(files1) if files1 else 0 @@ -2351,12 +2323,11 @@ def compare_md5sum(file1, file2, logact = 0): if fcnt1 != fcnt2: return 1 chksm1 = get_md5sum(files1, fcnt1, logact) chksm1 = ''.join(chksm1) - chksm2 = get_md5sum(files1, fcnt2, logact) + chksm2 = get_md5sum(files2, fcnt2, logact) chksm2 = ''.join(chksm2) else: chksm1 = get_md5sum(file1, 0, logact) chksm2 = get_md5sum(file2, 0, logact) - return (0 if (chksm1 and chksm2 and chksm1 == chksm2) else 1) # @@ -2392,13 +2363,11 @@ def change_local_directory(todir, logact = 0): # pass in empty dir to turn the recording delete directory on # def record_delete_directory(dir, val): - global DIRLVLS - if dir is None: if isinstance(val, int): DIRLVLS = val - elif re.match(r'^\d+$'): + elif val.isdigit(): DIRLVLS = int(val) elif dir and not re.match(r'^(\.|\./|/)$', dir) and dir not in DELDIRS: DELDIRS[dir] = val @@ -2407,9 +2376,7 @@ def record_delete_directory(dir, val): # remove the recorded delete directory if it is empty # def clean_delete_directory(logact = 0): - - global DIRLVLS, DELDIRS - + global DELDIRS, DIRLVLS if not DIRLVLS: return if logact: lact = logact&~(PgLOG.EXITLG) @@ -2417,8 +2384,8 @@ def clean_delete_directory(logact = 0): logact = lact = PgLOG.LOGWRN lvl = DIRLVLS DIRLVLS = 0 # set to 0 to stop recording directory - while lvl > 0: - lvl -= 1 + while lvl != 0: + if lvl > 0: lvl -= 1 dirs = {} for dir in DELDIRS: host = DELDIRS[dir] @@ -2430,12 +2397,10 @@ def clean_delete_directory(logact = 0): elif dstat > 0: if dstat == 1 and lvl > 0: PgLOG.pglog(dinfo + ": Directory not empty yet", lact) continue - - if lvl: dirs[op.dirname(dir)] = host - + pdir = op.dirname(dir) + if lvl and pdir and not re.match(r'^(\.|\./|/)$', pdir): dirs[pdir] = host if not dirs: break DELDIRS = dirs - DELDIRS = {} # empty cache afterward # @@ -2720,12 +2685,10 @@ def lmsg(file, msg, logact = 0): # return PgLOG.SUCCESS if yes PgLOG.FAILURE if not # def check_local_executable(path, actstr = '', logact = 0): - - if os.access(path, os.W_OK): return PgLOG.SUCCESS + if os.access(path, os.X_OK): return PgLOG.SUCCESS if check_local_accessible(path, actstr, logact): if actstr: actstr += '-' errlog("{}{}: Accessible, but Unexecutable on'{}'".format(actstr, path, PgLOG.PGLOG['HOSTNAME']), 'L', 1, logact) - return PgLOG.FAILURE @@ -2772,13 +2735,10 @@ def check_webfile_writable(action, wfile, logact = 0): # convert the one file to another via uncompress, move/copy, and/or compress # def convert_files(ofile, ifile, keep = 0, logact = 0): - if ofile == ifile: return PgLOG.SUCCESS oname = ofile iname = ifile - if keep: kfile = ifile + ".keep" - oext = iext = None for ext in PGCMPS: if oext is None: @@ -2791,41 +2751,35 @@ def convert_files(ofile, ifile, keep = 0, logact = 0): if ms: iname = ms.group(1) iext = ext - if iext and oext and oext == iext: oext = iext = None iname = ifile oname = ofile - if iext: # uncompress if keep: if iext == 'zip': kfile = ifile else: local_copy_local(kfile, ifile, logact) - if PgLOG.pgsystem("{} {}".format(PGCMPS[iext][1], ifile), logact, 5): if iext == "zip": path = op.dirname(iname) if path and path != '.': move_local_file(iname, op.basename(iname), logact) if not keep: delete_local_file(ifile, logact) - if oname != iname: # move/copy path = op.dirname(oname) if path and not op.exists(path): make_local_directory(path, logact) if keep and not op.exists(kfile): local_copy_local(oname, iname, logact) - kfile = iname + kfile = iname else: move_local_file(oname, iname, logact) - if oext: # compress if keep and not op.exists(kfile): if oext == "zip": kfile = oname else: local_copy_local(kfile, oname, logact) - if oext == "zip": path = op.dirname(oname) if path: @@ -2835,17 +2789,14 @@ def convert_files(ofile, ifile, keep = 0, logact = 0): if path != '.': change_local_directory(path, logact) else: PgLOG.pgsystem("{} {} {}".format(PGCMPS[oext][0], ofile, oname), logact, 5) - if not keep and op.exists(ofile): delete_local_file(oname, logact) else: PgLOG.pgsystem("{} {}".format(PGCMPS[oext][0], oname), logact, 5) - if keep and op.exists(kfile) and kfile != ifile: - if op.exist(ifile): + if op.exists(ifile): delete_local_file(kfile, logact) else: move_local_file(ifile, kfile, logact) - if op.exists(ofile): return PgLOG.SUCCESS else: @@ -2915,13 +2866,12 @@ def read_local_file(file, logact = 0): # # open a local file and return the file handler # -def open_local_file(file, mode = 'r', logact = PgLOG.LOGERR): - +def open_local_file(file, mode = 'r', logact = None): + if logact is None: logact = PgLOG.LOGERR try: fd = open(file, mode) except Exception as e: return errlog("{}: {}".format(file, str(e)), 'L', 1, logact) - return fd # diff --git a/src/rda_python_common/PgLOG.py b/src/rda_python_common/PgLOG.py index c71b899..b7e9bf9 100644 --- a/src/rda_python_common/PgLOG.py +++ b/src/rda_python_common/PgLOG.py @@ -21,6 +21,7 @@ import grp import shlex import smtplib +import subprocess from email.message import EmailMessage from subprocess import Popen, PIPE from os import path as op @@ -95,19 +96,14 @@ 'SETUID' : '', # the login name for suid if it is different to the CURUID 'FILEMODE': 0o664, # default 8-base file mode 'EXECMODE': 0o775, # default 8-base executable file mode or directory mode - 'ARCHHOST': "hpss", # change to hpss from mss - 'ARCHROOT': "/FS/DECS", # root path for segregated tape on hpss - 'BACKROOT': "/DRDATA/DECS", # backup path for desaster recovering tape on hpss - 'OLDAROOT': "/FS/DSS", # old root path on hpss - 'OLDBROOT': "/DRDATA/DSS", # old backup tape on hpss - # COMMONUSER and ADMINUSER are set below via SETPGLOG (env overrides PG) + # COMMONUSER and ADMINUSER are set in set_common_pglog() (env overrides) 'SUDOGDEX' : 0, # 1 to allow sudo to PGLOG['COMMONUSER'] 'HOSTNAME' : '', # current host name the process in running on 'OBJCTSTR' : "object", 'BACKUPNM' : "quasar", 'DRDATANM' : "drdata", + 'TACCNAME' : "tacc", 'GPFSNAME' : "glade", - 'SLMNAME' : "SLURM", 'PBSNAME' : "PBS", 'DSIDCHRS' : "d", 'DOSHELL' : False, @@ -116,16 +112,14 @@ 'BCHHOSTS' : "PBS", 'HOSTTYPE' : 'dav', # default HOSTTYPE 'EMLMAX' : 256, # up limit of email line count - 'PGBATCH' : '', # current batch service name, SLURM or PBS + 'PGBATCH' : '', # current batch service name, PBS 'PGBINDIR' : '', - 'SLMTIME' : 604800, # max runtime for SLURM bath job, (7x24x60x60 seconds) 'PBSTIME' : 86400, # max runtime for PBS bath job, (24x60x60 seconds) 'MSSGRP' : None, # set if set to different HPSS group 'GDEXGRP' : "decs", 'EMLSEND' : None, # path to sendmail, None if not exists 'DSCHECK' : None, # carry some cached dscheck information 'PGDBBUF' : None, # reference to a connected database object - 'HPSSLMT' : 10, # up limit of HPSS streams 'NOQUIT' : 0, # do not quit if this flag is set for daemons 'DBRETRY' : 2, # db retry count after error 'TIMEOUT' : 15, # default timeout (in seconds) for tosystem() @@ -139,30 +133,12 @@ 'EMLPORT' : 25 } -def SETPGLOG(key, default): - """Set ``PGLOG[key]`` from environment variable ``PG`` or fall back - to ``default`` if the variable is unset. Used to make per-environment - overrides (e.g. PGCOMMONUSER, PGADMINUSER) survive package upgrades.""" - PGLOG[key] = os.environ.get('PG' + key, default) - -SETPGLOG("COMMONUSER", "gdexdata") -SETPGLOG("ADMINUSER", "zji") - -PGLOG['RDAUSER'] = PGLOG['COMMONUSER'] -PGLOG['RDAGRP'] = PGLOG['GDEXGRP'] -PGLOG['RDAEMAIL'] = PGLOG['ADMINUSER'] -PGLOG['SUDORDA'] = PGLOG['SUDOGDEX'] -# backwards-compat aliases (deprecated: use COMMONUSER / ADMINUSER) -PGLOG['GDEXUSER'] = PGLOG['COMMONUSER'] -PGLOG['GDEXEMAIL'] = PGLOG['ADMINUSER'] - HOSTTYPES = { 'rda' : 'dsg_mach', + 'crlogin' : 'dav', 'casper' : 'dav', 'crhtc' : 'dav', 'cron' : 'dav', - 'cheyenne' : 'ch', - 'chadmin' : 'ch' } CPID = { @@ -172,25 +148,21 @@ def SETPGLOG(key, default): 'CPID' : "", } -BCHCMDS = {'PBS' : 'qsub', 'SLURM' : 'sbatch'} +BCHCMDS = {'PBS' : 'qsub'} # global dists to cashe information COMMANDS = {} -SLMHOSTS = [] -SLMSTATS = {} +CMDPATHS = {} # cache of bare command name -> full path (or '' if not found) PBSHOSTS = [] PBSSTATS = {} +OUTPUT = None # result output destination, opened via PgOPT.open_output() # # get time string in format YYMMDDHHNNSS for given ctime; or current time if ctime is 0 # -def current_datetime(ctime = 0): - - if PGLOG['GMTZ']: - dt = time.gmtime(ctime) if ctime else time.gmtime() - else: - dt = time.localtime(ctime) if ctime else time.localtime() - +def current_datetime(ctime=0): + get_time = time.gmtime if PGLOG['GMTZ'] else time.localtime + dt = get_time(ctime) if ctime else get_time() return "{:02}{:02}{:02}{:02}{:02}{:02}".format(dt[0], dt[1], dt[2], dt[3], dt[4], dt[5]) # @@ -207,46 +179,42 @@ def get_environment(name, default = None, logact = 0): # # cache the msg string to global email entries for later call of send_email() # -def set_email(msg, logact = 0): - +def set_email(msg, logact=0): if logact and msg: if logact&EMLTOP: if PGLOG['PRGMSG']: msg = PGLOG['PRGMSG'] + "\n" + msg PGLOG['PRGMSG'] = "" if PGLOG['ERRCNT'] == 0: - if not re.search(r'\n$', msg): msg += "!\n" + if not msg.endswith('\n'): msg += "!\n" else: if PGLOG['ERRCNT'] == 1: msg += " with 1 Error:\n" else: msg += " with {} Errors:\n".format(PGLOG['ERRCNT']) + msg += "\nERROR MESSAGE:\n" msg += break_long_string(PGLOG['ERRMSG'], 512, None, PGLOG['EMLMAX']/2, None, 50, 25) PGLOG['ERRCNT'] = 0 PGLOG['ERRMSG'] = '' - if PGLOG['SUMMSG']: - msg += PGLOG['SEPLINE'] - if PGLOG['SUMMSG']: msg += "Summary:\n" + msg += "\nSUMMARY:\n" msg += break_long_string(PGLOG['SUMMSG'], 512, None, PGLOG['EMLMAX']/2, None, 50, 25) - if PGLOG['EMLMSG']: - msg += PGLOG['SEPLINE'] - if PGLOG['SUMMSG']: msg += "Detail Information:\n" - + msg += "\nDETAIL INFORMATION:\n" PGLOG['EMLMSG'] = msg + break_long_string(PGLOG['EMLMSG'], 512, None, PGLOG['EMLMAX'], None, 50, 40) PGLOG['SUMMSG'] = "" # in case not else: if logact&ERRLOG: # record error for email summary PGLOG['ERRCNT'] += 1 - if logact&BRKLIN: PGLOG['ERRMSG'] += "\n" + if PGLOG['ERRMSG']: # blank line between consecutive errors + if not PGLOG['ERRMSG'].endswith('\n'): PGLOG['ERRMSG'] += "\n" + PGLOG['ERRMSG'] += "\n" PGLOG['ERRMSG'] += "{}. {}".format(PGLOG['ERRCNT'], msg) elif logact&EMLSUM: if PGLOG['SUMMSG']: if logact&BRKLIN: PGLOG['SUMMSG'] += "\n" if logact&SEPLIN: PGLOG['SUMMSG'] += PGLOG['SEPLINE'] PGLOG['SUMMSG'] += msg # append - if logact&EMLLOG: if PGLOG['EMLMSG']: if logact&BRKLIN: PGLOG['EMLMSG'] += "\n" @@ -264,15 +232,14 @@ def get_email(): # # send a customized email with all entries included # -def send_customized_email(logmsg, emlmsg, logact = LOGWRN): - +def send_customized_email(logmsg, emlmsg, logact=None): + if logact is None: logact = LOGWRN entries = { - 'fr' : ["From", 1, None], - 'to' : ["To", 1, None], - 'cc' : ["Cc", 0, ''], - 'sb' : ["Subject", 1, None] + 'fr': ["From", 1, None], + 'to': ["To", 1, None], + 'cc': ["Cc", 0, ''], + 'sb': ["Subject", 1, None] } - if logmsg: logmsg += ': ' else: @@ -287,10 +254,8 @@ def send_customized_email(logmsg, emlmsg, logact = LOGWRN): if vals[2]: entries[ekey][2] = vals[2] elif entries[ekey][1]: return pglog("{}Missing Entry '{}' for sending email".format(logmsg, entry), logact|ERRLOG) - ret = send_python_email(entries['sb'][2], entries['to'][2], msg, entries['fr'][2], entries['cc'][2], logact) - if ret == SUCCESS or not PGLOG['EMLSEND']: return ret - + if ret == SUCCESS or not PGLOG['EMLSEND']: return ret # try commandline sendmail ret = pgsystem(PGLOG['EMLSEND'], logact, 4, emlmsg) logmsg += "Email " + entries['to'][2] @@ -302,29 +267,27 @@ def send_customized_email(logmsg, emlmsg, logact = LOGWRN): else: errmsg = "Error sending email: " + logmsg pglog(errmsg, (logact|ERRLOG)&~EXITLG) - return ret # # send an email; if empty msg send email message saved in PGLOG['EMLMSG'] instead # -def send_email(subject = None, receiver = None, msg = None, sender = None, logact = LOGWRN): - +def send_email(subject=None, receiver=None, msg=None, sender=None, logact=None): + if logact is None: logact = LOGWRN return send_python_email(subject, receiver, msg, sender, None, logact) # # send an email via python module smtplib; if empty msg send email message saved # in PGLOG['EMLMSG'] instead. pass cc = '' for skipping 'Cc: ' # -def send_python_email(subject = None, receiver = None, msg = None, sender = None, cc = None, logact = LOGWRN): - +def send_python_email(subject=None, receiver=None, msg=None, sender=None, cc=None, logact=None): + if logact is None: logact = LOGWRN if not msg: if PGLOG['EMLMSG']: msg = PGLOG['EMLMSG'] PGLOG['EMLMSG'] = '' else: return '' - docc = False if cc else True if not sender: sender = PGLOG['CURUID'] @@ -335,7 +298,6 @@ def send_python_email(subject = None, receiver = None, msg = None, sender = None receiver = PGLOG['EMLADDR'] if PGLOG['EMLADDR'] else PGLOG['CURUID'] if receiver == PGLOG['COMMONUSER']: receiver = PGLOG['ADMINUSER'] if receiver.find('@') == -1: receiver += "@ucar.edu" - if docc and not re.match(PGLOG['COMMONUSER'], sender): add_carbon_copy(sender, 1) emlmsg = EmailMessage() emlmsg.set_content(msg) @@ -351,6 +313,7 @@ def send_python_email(subject = None, receiver = None, msg = None, sender = None emlmsg['Subject'] = subject if CPID['CPID']: logmsg += " in " + CPID['CPID'] logmsg += ", Subject: {}\n".format(subject) + eml = None try: eml = smtplib.SMTP(PGLOG['EMLSRVR'], PGLOG['EMLPORT']) eml.send_message(emlmsg) @@ -358,25 +321,25 @@ def send_python_email(subject = None, receiver = None, msg = None, sender = None errmsg = f"Error sending email:\n{err}\n{logmsg}" return pglog(errmsg, (logact|ERRLOG)&~EXITLG) finally: - eml.quit() - log_email(str(emlmsg)) - pglog(logmsg, logact&~EXITLG) - return SUCCESS + if eml is not None: + eml.quit() + log_email(str(emlmsg)) + pglog(logmsg, logact&~EXITLG) + return SUCCESS # # log email sent # def log_email(emlmsg): - - if not CPID['PID']: CPID['PID'] = "{}-{}-{}".format(PGLOG['HOSTNAME'], get_command(), PGLOG['CURUID']) + if not CPID['PID']: + CPID['PID'] = "{}-{}-{}".format(PGLOG['HOSTNAME'], get_command(), PGLOG['CURUID']) cmdstr = "{} {} at {}\n".format(CPID['PID'], break_long_string(CPID['CMD'], 40, "...", 1), current_datetime()) fn = "{}/{}".format(PGLOG['LOGPATH'], PGLOG['EMLFILE']) try: - f = open(fn, 'a') - f.write(cmdstr + emlmsg) - f.close() + with open(fn, 'a') as f: + f.write(cmdstr + emlmsg) except FileNotFoundError as e: - print(e) + print(e) # # Function: cmdlog(cmdline) @@ -416,8 +379,8 @@ def cmdlog(cmdline = None, ctime = 0, logact = None): # # log and display message/error and exit program according logact value # -def pglog(msg, logact = MSGLOG): - +def pglog(msg, logact=None): + if logact is None: logact = MSGLOG retmsg = None logact &= PGLOG['LOGMASK'] # filtering the log actions if logact&RCDMSG: logact |= MSGLOG @@ -425,16 +388,14 @@ def pglog(msg, logact = MSGLOG): if logact&EMEROL: if logact&EMLLOG: logact &= ~EMLLOG if not logact&ERRLOG: logact &= ~EMEROL - msg = msg.lstrip() if msg else '' # remove leading whitespaces for logging message if logact&EXITLG: ext = "Exit 1 in {}\n".format(os.getcwd()) if msg: msg = msg.rstrip() + "; " msg += ext else: - if msg and not re.search(r'(\n|\r)$', msg): msg += "\n" + if msg and not msg.endswith(('\n', '\r')): msg += "\n" if logact&RETMSG: retmsg = msg - if logact&EMLALL: if logact&SNDEML or not msg: title = (msg if msg else "Message from {}-{}".format(PGLOG['HOSTNAME'], get_command())) @@ -442,22 +403,18 @@ def pglog(msg, logact = MSGLOG): send_email(title.rstrip()) elif msg: set_email(msg, logact) - if not msg: return (retmsg if retmsg else FAILURE) - if logact&EXITLG and (PGLOG['EMLMSG'] or PGLOG['SUMMSG'] or PGLOG['ERRMSG'] or PGLOG['PRGMSG']): if not logact&EMLALL: set_email(msg, logact) title = "ABORTS {}-{}".format(PGLOG['HOSTNAME'], get_command()) set_email((("ABORTS " + CPID['PID']) if CPID['PID'] else title), EMLTOP) msg = title + '\n' + msg - send_email(title) - + send_email(title) if logact&LOGERR: # make sure error is always logged msg = break_long_string(msg) if logact&(ERRLOG|EXITLG): cmdstr = get_error_command(int(time.time()), logact) msg = cmdstr + msg - if not logact&NOTLOG: if logact&ERRLOG: if not PGLOG['ERRFILE']: PGLOG['ERRFILE'] = re.sub(r'.log$', '.err', PGLOG['LOGFILE']) @@ -466,7 +423,6 @@ def pglog(msg, logact = MSGLOG): write_message(cmdstr, f"{PGLOG['LOGPATH']}/{PGLOG['LOGFILE']}", logact) else: write_message(msg, f"{PGLOG['LOGPATH']}/{PGLOG['LOGFILE']}", logact) - if not PGLOG['BCKGRND'] and logact&(ERRLOG|WARNLG): write_message(msg, None, logact) @@ -500,9 +456,9 @@ def write_message(msg, file, logact): # # check and disconnet database before exit # -def pgexit(stat = 0): - +def pgexit(stat=0): if PGLOG['PGDBBUF']: PGLOG['PGDBBUF'].close() + if OUTPUT and OUTPUT != sys.stdout: OUTPUT.close() sys.exit(stat) # @@ -521,18 +477,16 @@ def get_error_command(ctime, logact): # # get call trace track # -def get_call_trace(cut = 1): - +def get_call_trace(cut=1): t = traceback.extract_stack() n = len(t) - cut - str = '' + trace = '' sep = 'Trace: ' for i in range(n): - tc = t[i] - str += "{}{}({}){}".format(sep, tc[0], tc[1], ("" if tc[2] == '' else "{%s()}" % tc[2])) - if i == 0: sep = '=>' - - return str + "\n" if str else "" + tc = t[i] + trace += "{}{}({}){}".format(sep, tc[0], tc[1], ("" if tc[2] == '' else "{%s()}" % tc[2])) + if i == 0: sep = '=>' + return trace + "\n" if trace else "" # # get caller file name @@ -544,14 +498,11 @@ def get_caller_file(cidx = 0): # # log message, msg, for degugging processes according to the debug level # -def pgdbg(level, msg = None, do_trace = True): - +def pgdbg(level, msg=None, do_trace=True): if not PGLOG['DBGLEVEL']: return # no further action - if not isinstance(level, int): ms = re.match(r'^(\d+)', level) level = int(ms.group(1)) if ms else 0 - levels = [0, 0] if isinstance(PGLOG['DBGLEVEL'], int): levels[1] = PGLOG['DBGLEVEL'] @@ -564,9 +515,7 @@ def pgdbg(level, msg = None, do_trace = True): if ms: levels[0] = int(ms.group(1)) if ms.group(1) else 0 levels[1] = int(ms.group(2)) if ms.group(2) else 9999 - if level > levels[1] or level < levels[0]: return # debug level is out of range - if 'DBGPATH' in PGLOG: dfile = PGLOG['DBGPATH'] + '/' + PGLOG['DBGFILE'] else: @@ -576,12 +525,10 @@ def pgdbg(level, msg = None, do_trace = True): msg = "DEBUG for " + CPID['PID'] + " " if CPID['CPID']: msg += CPID['CPID'] + " <= " msg += break_long_string(CPID['CMD'], 40, "...", 1) - # logging debug info - DBG = open(dfile, 'a') - DBG.write("{}:{}\n".format(level, msg)) - if do_trace: DBG.write(get_call_trace()) - DBG.close() + with open(dfile, 'a') as DBG: + DBG.write("{}:{}\n".format(level, msg)) + if do_trace: DBG.write(get_call_trace()) # # return trimed string (strip leading and trailling spaces); remove comments led by '#' if rmcmt > 0 @@ -616,13 +563,10 @@ def set_help_path(progfile): # show program usage in file "PGLOG['PUSGDIR']/progname.usg" on screen with unix # system function 'pg', exit program when done. # -def show_usage(progname, opts = None): - +def show_usage(progname, opts=None): if PGLOG['PUSGDIR'] is None: set_help_path(get_caller_file(1)) usgname = join_paths(PGLOG['PUSGDIR'], progname + '.usg') - - if opts: - # show usage for individual option of dsarch + if opts: # show usage for individual option of dsarch for opt in opts: if opts[opt][0] == 0: msg = "Mode" @@ -632,26 +576,23 @@ def show_usage(progname, opts = None): msg = "Multi-Value Information" else: msg = "Action" - sys.stdout.write("\nDescription of {} Option -{}:\n".format(msg, opt)) - IN = open(usgname, 'r') nilcnt = begin = 0 - for line in IN: - if begin == 0: - rx = " -{} or -".format(opt) - if re.match(rx, line): begin = 1 - elif re.match(r'^\s*$', line): - if nilcnt: break - nilcnt = 1 - else: - if re.match(r'\d[\.\s\d]', line): break # section title - if nilcnt and re.match(r' -\w\w or -', line): break - nilcnt = 0 - if begin: sys.stdout.write(line) - IN.close() + with open(usgname, 'r') as IN: + for line in IN: + if begin == 0: + rx = " -{} or -".format(opt) + if re.match(rx, line): begin = 1 + elif re.match(r'^\s*$', line): + if nilcnt: break + nilcnt = 1 + else: + if re.match(r'\d[\.\s\d]', line): break # section title + if nilcnt and re.match(r' -\w\w or -', line): break + nilcnt = 0 + if begin: sys.stdout.write(line) else: - os.system("more " + usgname) - + subprocess.run(['more', usgname]) pgexit(0) # @@ -695,28 +636,24 @@ def std2err(line): # instr - input string passing to the command via stdin if not None # seconds - number of seconds to wait for a timeout process if > 0 # -def pgsystem(pgcmd, logact = LOGWRN, cmdopt = 5, instr = None, seconds = 0): - +def pgsystem(pgcmd, logact=None, cmdopt=5, instr=None, seconds=0): + if logact is None: logact = LOGWRN ret = SUCCESS if not pgcmd: return ret # empty command - act = logact&~EXITLG if act&ERRLOG: act &= ~ERRLOG act |= WARNLG - if act&MSGLOG: act |= FRCLOG # make sure system calls always logged cmdact = act if cmdopt&1 else 0 doshell = True if cmdopt&1024 else PGLOG['DOSHELL'] - if isinstance(pgcmd, str): cmdstr = pgcmd if not doshell and re.search(r'[*?<>|;]', pgcmd): doshell = True execmd = pgcmd if doshell else shlex.split(pgcmd) else: cmdstr = shlex.join(pgcmd) - execmd = cmdstr if doshell else pgcmd - + execmd = cmdstr if doshell else pgcmd if cmdact: if cmdopt&8: cmdlog("starts '{}'".format(cmdstr), None, cmdact) @@ -757,13 +694,11 @@ def pgsystem(pgcmd, logact = LOGWRN, cmdopt = 5, instr = None, seconds = 0): else: ret = FAILURE if FD.returncode else SUCCESS if isinstance(outbuf, bytes): outbuf = str(outbuf, errors='replace') - if isinstance(errbuf, bytes): errbuf = str(errbuf, errors='replace') - + if isinstance(errbuf, bytes): errbuf = str(errbuf, errors='replace') if errbuf and cmdopt&32: outbuf += errbuf if cmdopt&256: PGLOG['SYSERR'] = errbuf errbuf = '' - if outbuf: lines = outbuf.split('\n') for line in lines: @@ -779,7 +714,6 @@ def pgsystem(pgcmd, logact = LOGWRN, cmdopt = 5, instr = None, seconds = 0): elif stdlog: pglog(line, stdlog) if cmdopt&16: retbuf += line + "\n" - if errbuf: lines = errbuf.split('\n') for line in lines: @@ -791,36 +725,30 @@ def pgsystem(pgcmd, logact = LOGWRN, cmdopt = 5, instr = None, seconds = 0): else: if cmdopt&260: error += line + "\n" if abort == -1 and re.match('ABORTS ', line): abort = 1 - if ret == SUCCESS and abort == 1: ret = FAILURE end = time.time() last = end - last - if error: + cmdpstr = command_path(cmdstr) if ret == FAILURE: - error = "Error Execute: {}\n{}".format(cmdstr, error) + error = "Error Execute: {}\n{}".format(cmdpstr, error) else: - error = "Error From: {}\n{}".format(cmdstr, error) - + error = "Error From: {}\n{}".format(cmdpstr, error) if loop > 1: error = "Retry " if cmdopt&256: PGLOG['SYSERR'] += error if cmdopt&4: errlog = (act|ERRLOG) if ret == FAILURE and loop >= loops: errlog |= logact pglog(error, errlog) - if last > PGLOG['CMDTIME'] and not re.search(r'(^|/|\s)(dsarch|dsupdt|dsrqst)\s', cmdstr): cmdstr = "> {} Ends By {}".format(break_long_string(cmdstr, 100, "...", 1), current_datetime()) cmd_execute_time(cmdstr, last, cmdact) - if ret == SUCCESS or loop >= loops: break time.sleep(6) - if ret == FAILURE and retbuf and cmdopt&272 == 272: if PGLOG['SYSERR']: PGLOG['SYSERR'] += '\n' PGLOG['SYSERR'] += retbuf retbuf = '' - return (retbuf if cmdopt&16 else ret) # @@ -854,35 +782,27 @@ def cmd_execute_time(cmdstr, last, logact = None): # # convert given seconds to string time with units of S-Second,M-Minute,H-Hour,D-Day # -def seconds_to_string_time(seconds, showzero = 0): - +def seconds_to_string_time(seconds, showzero=0): msg = '' - s = m = h = 0 - if seconds > 0: - s = seconds%60 # seconds (0-59) - minutes = int(seconds/60) # total minutes - m = minutes%60 # minutes (0-59) - if minutes >= 60: - hours = int(minutes/60) # total hours - h = hours%24 # hours (0-23) - if hours >= 24: - msg += "{}D".format(int(hours/24)) # days - if h: msg += "{}H".format(h) - if m: msg += "{}M".format(m) + minutes, s = divmod(seconds, 60) + hours, m = divmod(int(minutes), 60) + days, h = divmod(hours, 24) + if days: msg += "{}D".format(days) + if h: msg += "{}H".format(h) + if m: msg += "{}M".format(int(m)) if s: - msg += "%dS"%(s) if isinstance(s, int) else "{:.3f}S".format(s) + msg += "%dS" % s if isinstance(s, int) else "{:.3f}S".format(s) elif showzero: msg = "0S" - return msg # # wrap function to call pgsystem() with a timeout control # return FAILURE if error eval or time out # -def tosystem(cmd, timeout = 0, logact = LOGWRN, cmdopt = 5, instr = None): - +def tosystem(cmd, timeout=0, logact=None, cmdopt=5, instr=None): + if logact is None: logact = LOGWRN if not timeout: timeout = PGLOG['TIMEOUT'] # set default timeout if missed return pgsystem(cmd, logact, cmdopt, instr, timeout) @@ -1018,8 +938,7 @@ def valid_batch_host(host, logact = 0): # # Return the full command path if valid; '' if not # -def valid_command(cmd, logact = 0): - +def valid_command(cmd, logact=0): ms = re.match(r'^(\S+)( .*)$', cmd) if ms: option = ms.group(2) @@ -1029,14 +948,30 @@ def valid_command(cmd, logact = 0): if cmd not in COMMANDS: buf = shutil.which(cmd) if buf is None: - if logact: pglog(cmd + ": executable command not found", logact) + if logact: pglog("{}: executable command not found in\n{}".format(cmd, os.environ.get("PATH")), logact) buf = '' elif option: buf += option COMMANDS[cmd] = buf - return COMMANDS[cmd] +# +# expand the leading bare command name of a command string to its full path +# +# Return the command string unchanged if the command cannot be located +# +def command_path(cmdstr): + if not cmdstr: return '' + sp = cmdstr.find(' ') + cmd = cmdstr if sp < 0 else cmdstr[:sp] + if '/' in cmd or '\\' in cmd: return cmdstr + pcmd = CMDPATHS.get(cmd) + if pcmd is None: + pcmd = shutil.which(cmd) or '' + CMDPATHS[cmd] = pcmd + if not pcmd: return cmdstr + return pcmd if sp < 0 else pcmd + cmdstr[sp:] + # # add carbon copies to PGLOG['CCDADDR'] # @@ -1088,62 +1023,24 @@ def get_short_host(host): return host -# -# get a live SLURM host name -# -def get_slurm_host(): - - global SLMHOSTS - - if not SLMSTATS and PGLOG['SLMHOSTS']: - SLMHOSTS = PGLOG['SLMHOSTS'].split(':') - for host in SLMHOSTS: - SLMSTATS[host] = 1 - - for host in SLMHOSTS: - if host in SLMSTATS and SLMSTATS[host]: return host - - return None - # # get a live PBS host name # def get_pbs_host(): - global PBSHOSTS - if not PBSSTATS and PGLOG['PBSHOSTS']: PBSHOSTS = PGLOG['PBSHOSTS'].split(':') for host in PBSHOSTS: PBSSTATS[host] = 1 - for host in PBSHOSTS: if host in PBSSTATS and PBSSTATS[host]: return host - return None -# -# set host status, 0 dead & 1 live, for one or all avalaible slurm hosts -# -def set_slurm_host(host = None, stat = 0): - - global SLMHOSTS - - if host: - SLMSTATS[host] = stat - else: - if not SLMHOSTS and PGLOG['SLMHOSTS']: - SLMHOSTS = PGLOG['SLMHOSTS'].split(':') - for host in SLMHOSTS: - SLMSTATS[host] = stat - # # set host status, 0 dead & 1 live, for one or all avalaible pbs hosts # -def set_pbs_host(host = None, stat = 0): - +def set_pbs_host(host=None, stat=0): global PBSHOSTS - if host: PBSSTATS[host] = stat else: @@ -1155,17 +1052,16 @@ def set_pbs_host(host = None, stat = 0): # # reset the batch host name in case was not set properly # -def reset_batch_host(bhost, logact = LOGWRN): - - BCHHOST = bhost.upper() - - if BCHHOST != PGLOG['PGBATCH']: +def reset_batch_host(bhost, logact=None): + if logact is None: logact = LOGWRN + bchhost = bhost.upper() + if bchhost != PGLOG['PGBATCH']: if PGLOG['CURBID'] > 0: - pglog("{}-{}: Batch ID is set, cannot change Batch host to {}".format(PGLOG['PGBATCH'], PGLOG['CURBID'], BCHHOST) , logact) + pglog("{}-{}: Batch ID is set, cannot change Batch host to {}".format(PGLOG['PGBATCH'], PGLOG['CURBID'], bchhost) , logact) else: - ms = re.search(r'(^|:){}(:|$)'.format(BCHHOST), PGLOG['BCHHOSTS']) + ms = re.search(r'(^|:){}(:|$)'.format(bchhost), PGLOG['BCHHOSTS']) if ms: - PGLOG['PGBATCH'] = BCHHOST + PGLOG['PGBATCH'] = bchhost if PGLOG['CURBID'] == 0: PGLOG['CURBID'] = -1 elif PGLOG['PGBATCH']: PGLOG['PGBATCH'] = '' @@ -1214,31 +1110,6 @@ def get_remote_command(cmd, host, asuser = None): # if host and not re.match(PGLOG['HOSTNAME'], host): cmd = "ssh {} {}".format(host, cmd) return get_local_command(cmd, asuser) -# -# wrap a given hpss command cmd with sudo either before his of after hsi -# to run as user asuser -# -def get_hpss_command(cmd, asuser = None, hcmd = None): - - cuser = PGLOG['SETUID'] if PGLOG['SETUID'] else PGLOG['CURUID'] - if not hcmd: hcmd = 'hsi' - - if asuser and cuser != asuser: - if cuser == PGLOG['COMMONUSER']: - return "{} sudo -u {} {}".format(hcmd, asuser, cmd) # setuid wrapper as user asuser - elif PGLOG['SUDOGDEX'] and asuser == PGLOG['COMMONUSER']: - return "sudo -u {} {} {}".format(PGLOG['COMMONUSER'], hcmd, cmd) # sudo as user gdexdata - - if cuser != PGLOG['COMMONUSER']: - if re.match(r'^ls ', cmd) and hcmd == 'hsi': - return "hpss" + cmd # use 'hpssls' instead of 'hsi ls' - elif re.match(r'^htar -tvf', hcmd): - hcmd.replace('htar -tvf', 'htarmember', 1) # use 'htarmember' instead of 'htar -tvf' - elif re.match(r'^hsi ls', hcmd): - hcmd.replce('hsi ls', 'hpssls', 1) # use 'hpssls' instead of 'hsi ls' - - return "{} {}".format(hcmd, cmd) - # # wrap a given sync command for given host name with/without sudo # @@ -1269,9 +1140,17 @@ def set_suid(cuid = 0): # set comman pglog # def set_common_pglog(): - + # resolve common/admin user from environment (COMMONUSER / ADMINUSER) + SETPGLOG("COMMONUSER", "gdexdata") + SETPGLOG("ADMINUSER", "zji") + PGLOG['RDAUSER'] = PGLOG['COMMONUSER'] + PGLOG['RDAGRP'] = PGLOG['GDEXGRP'] + PGLOG['RDAEMAIL'] = PGLOG['ADMINUSER'] + PGLOG['SUDORDA'] = PGLOG['SUDOGDEX'] + # backwards-compat aliases (deprecated: use COMMONUSER / ADMINUSER) + PGLOG['GDEXUSER'] = PGLOG['COMMONUSER'] + PGLOG['GDEXEMAIL'] = PGLOG['ADMINUSER'] PGLOG['CURDIR'] = os.getcwd() - # set current user id PGLOG['RUID'] = os.getuid() PGLOG['EUID'] = os.geteuid() @@ -1279,11 +1158,10 @@ def set_common_pglog(): try: PGLOG['RDAUID'] = PGLOG['GDEXUID'] = pwd.getpwnam(PGLOG['COMMONUSER']).pw_uid PGLOG['RDAGID'] = PGLOG['GDEXGID'] = grp.getgrnam(PGLOG['GDEXGRP']).gr_gid - except: + except KeyError: PGLOG['RDAUID'] = PGLOG['GDEXUID'] = 0 PGLOG['RDAGID'] = PGLOG['GDEXGID'] = 0 - if PGLOG['CURUID'] == PGLOG['COMMONUSER']: PGLOG['SETUID'] = PGLOG['COMMONUSER'] - + if PGLOG['CURUID'] == PGLOG['COMMONUSER']: PGLOG['SETUID'] = PGLOG['COMMONUSER'] PGLOG['HOSTNAME'] = get_host() for htype in HOSTTYPES: ms = re.match(r'^{}(-|\d|$)'.format(htype), PGLOG['HOSTNAME']) @@ -1291,26 +1169,18 @@ def set_common_pglog(): PGLOG['HOSTTYPE'] = HOSTTYPES[htype] break PGLOG['DEFDSID'] = 'd000000' if PGLOG['NEWDSID'] else 'ds000.0' - PGLOG['NOTAROOT'] = '|'.join([PGLOG['OLDAROOT'], PGLOG['OLDBROOT'], PGLOG['BACKROOT']]) - PGLOG['NOTBROOT'] = '|'.join([PGLOG['OLDAROOT'], PGLOG['OLDBROOT'], PGLOG['ARCHROOT']]) - PGLOG['ALLROOTS'] = '|'.join([PGLOG['OLDAROOT'], PGLOG['OLDBROOT'], PGLOG['ARCHROOT'], PGLOG['BACKROOT']]) SETPGLOG("USRHOME", "/glade/u/home") SETPGLOG("DSSHOME", "/glade/u/home/gdexdata") SETPGLOG("GDEXHOME", "/data/local") SETPGLOG("ADDPATH", "") SETPGLOG("ADDLIB", "") SETPGLOG("OTHPATH", "") - SETPGLOG("PSQLHOME", "/usr/pgsql-15") + SETPGLOG("PSQLHOME", "") SETPGLOG("DSGHOSTS", "") SETPGLOG("DSIDCHRS", "d") - if not os.getenv('HOME'): os.environ['HOME'] = "{}/{}".format(PGLOG['USRHOME'], PGLOG['CURUID']) SETPGLOG("HOMEBIN", os.environ.get('HOME') + "/bin") - - if 'SLURM_JOBID' in os.environ: - PGLOG['CURBID'] = int(os.getenv('SLURM_JOBID')) - PGLOG['PGBATCH'] = PGLOG['SLMNAME'] - elif 'PBS_JOBID' in os.environ: + if 'PBS_JOBID' in os.environ: sbid = os.getenv('PBS_JOBID') ms = re.match(r'^(\d+)', sbid) PGLOG['CURBID'] = int(ms.group(1)) if ms else -1 @@ -1318,7 +1188,6 @@ def set_common_pglog(): else: PGLOG['CURBID'] = 0 PGLOG['PGBATCH'] = '' - pgpath = PGLOG['HOMEBIN'] PGLOG['LOCHOME'] = "/ncar/gdex/setuid" if not op.isdir(PGLOG['LOCHOME']): PGLOG['LOCHOME'] = "/usr/local/decs" @@ -1335,7 +1204,6 @@ def set_common_pglog(): pgpath = add_local_path(PGLOG['OTHPATH'], pgpath, 1) if PGLOG['ADDPATH']: pgpath = add_local_path(PGLOG['ADDPATH'], pgpath, 1) pgpath = add_local_path("/bin:/usr/bin:/usr/local/bin:/usr/sbin", pgpath, 1) - os.environ['PATH'] = pgpath os.environ['SHELL'] = '/bin/sh' # set PGLOG values with environments and defaults @@ -1347,11 +1215,12 @@ def set_common_pglog(): sm = "/usr/sbin/sendmail" if valid_command(sm): SETPGLOG("EMLSEND", f"{sm} -t") # send email command SETPGLOG("DBGLEVEL", '') # debug level - SETPGLOG("BAOTOKEN", 's.lh2t2kDjrqs3V8y2BU2zOocT') # OpenBao token - SETPGLOG("DBGPATH", PGLOG['DSSDBHM']+"/log") # path to debug log file - SETPGLOG("OBJCTBKT", "gdex-data") # default Bucket on Object Store - SETPGLOG("BACKUPEP", "gdex-quasar") # default Globus Endpoint on Quasar - SETPGLOG("DRDATAEP", "gdex-quasar-drdata") # DRDATA Globus Endpoint on Quasar + SETPGLOG("BAOTOKEN", 's.MdOPGayn0HcuuSPrmMqCvzJA') # OpenBao token + SETPGLOG("DBGPATH", PGLOG['DSSDBHM']+"/log") # path to debug log file + SETPGLOG("OBJCTBKT", "gdex-data") # default Bucket on Object Store + SETPGLOG("BACKUPEP", "gdex-quasar") # default Globus Endpoint on Quasar + SETPGLOG("DRDATAEP", "gdex-quasar-drdata") # DRDATA Globus Endpoint on Quasar + SETPGLOG("TACCEP", "gdex-tacc") # default Globus Endpoint on TACC SETPGLOG("DBGFILE", "pgdss.dbg") # debug file name SETPGLOG("CNFPATH", PGLOG['DSSHOME']+"/config") # path to configuration files SETPGLOG("DSSURL", "https://gdex.ucar.edu") # current dss web URL @@ -1360,48 +1229,40 @@ def set_common_pglog(): PGLOG['WEBHOSTS'] = PGLOG['WEBSERVERS'].split(':') if PGLOG['WEBSERVERS'] else [] SETPGLOG("DBMODULE", '') SETPGLOG("LOCDATA", "/data") - # set dss web homedir SETPGLOG("DSSWEB", PGLOG['LOCDATA']+"/web") SETPGLOG("DSWHOME", PGLOG['DSSWEB']+"/datasets") # datast web root path PGLOG['HOMEROOTS'] = "{}|{}".format(PGLOG['DSSHOME'], PGLOG['DSWHOME']) - SETPGLOG("DSSDATA", "/glade/campaign/collections/gdex") # dss data root path + SETPGLOG("DSSDATA", "/glade/campaign/collections/gdex") # dss data root path SETPGLOG("DSDHOME", PGLOG['DSSDATA']+"/data") # dataset data root path SETPGLOG("DECSHOME", PGLOG['DSSDATA']+"/decsdata") # dataset decsdata root path SETPGLOG("DSHHOME", PGLOG['DECSHOME']+"/helpfiles") # dataset help root path - SETPGLOG("GDEXWORK", "/lustre/desc1/gdex/work") # gdex work path + SETPGLOG("GDEXWORK", "/lustre/desc1/gdex/work") # gdex work path SETPGLOG("UPDTWKP", PGLOG['GDEXWORK']) # dsupdt work root path - SETPGLOG("TRANSFER", "/lustre/desc1/gdex/transfer") # gdex transfer path + SETPGLOG("TRANSFER", "/lustre/desc1/gdex/transfer") # gdex transfer path SETPGLOG("RQSTHOME", PGLOG['TRANSFER']+"/dsrqst") # dsrqst home - SETPGLOG("DSAHOME", "") # dataset data alternate root path - SETPGLOG("RQSTALTH", "") # alternate dsrqst path - SETPGLOG("GPFSHOST", "") # empty if writable to glade - SETPGLOG("PSQLHOST", "rda-db.ucar.edu") # host name for postgresql server - SETPGLOG("SLMHOSTS", "cheyenne:casper") # host names for SLURM server - SETPGLOG("PBSHOSTS", "cron:casper") # host names for PBS server - SETPGLOG("CHKHOSTS", "") # host names for dscheck daemon + SETPGLOG("DSAHOME", "") # dataset data alternate root path + SETPGLOG("RQSTALTH", "") # alternate dsrqst path + SETPGLOG("GPFSHOST", "") # empty if writable to glade + SETPGLOG("PSQLHOST", "rda-db.ucar.edu") # host name for postgresql server + SETPGLOG("PBSHOSTS", "cron:casper:crlogin") # host names for PBS server + SETPGLOG("CHKHOSTS", "") # host names for dscheck daemon SETPGLOG("PVIEWHOST", "pgdb02.k8s.ucar.edu") # host name for view only postgresql server SETPGLOG("PMISCHOST", "pgdb03.k8s.ucar.edu") # host name for misc postgresql server - SETPGLOG("FTPUPLD", PGLOG['TRANSFER']+"/rossby") # ftp upload path + SETPGLOG("FTPUPLD", PGLOG['TRANSFER']+"/rossby") # ftp upload path PGLOG['GPFSROOTS'] = "{}|{}|{}".format(PGLOG['DSDHOME'], PGLOG['UPDTWKP'], PGLOG['RQSTHOME']) - if 'ECCODES_DEFINITION_PATH' not in os.environ: os.environ['ECCODES_DEFINITION_PATH'] = "/usr/local/share/eccodes/definitions" os.environ['history'] = '0' - # set tmp dir SETPGLOG("TMPPATH", PGLOG['GDEXWORK'] + "/ptmp") if not PGLOG['TMPPATH']: PGLOG['TMPPATH'] = "/data/ptmp" - SETPGLOG("TMPDIR", '') if not PGLOG['TMPDIR']: PGLOG['TMPDIR'] = "/lustre/desc1/scratch/" + PGLOG['CURUID'] os.environ['TMPDIR'] = PGLOG['TMPDIR'] - # empty diretory for HOST-sync - - PGLOG['TMPSYNC'] = PGLOG['DSSDBHM'] + "/tmp/.syncdir" - + PGLOG['TMPSYNC'] = PGLOG['DSSDBHM'] + "/tmp/.syncdir" os.umask(2) # @@ -1447,42 +1308,33 @@ def SETPGLOG(name, value = ''): # set specialist home and return the default shell # def set_specialist_home(specialist): - if specialist == PGLOG['CURUID']: return # no need reset if 'MAIL' in os.environ and re.search(PGLOG['CURUID'], os.environ['MAIL']): - os.environ['MAIL'] = re.sub(PGLOG['CURUID'], specialist, os.environ['MAIL']) - + os.environ['MAIL'] = re.sub(PGLOG['CURUID'], specialist, os.environ['MAIL']) home = "{}/{}".format(PGLOG['USRHOME'], specialist) shell = "tcsh" - buf = pgsystem("grep ^{}: /etc/passwd".format(specialist), LOGWRN, 20) - if buf: - lines = buf.split('\n') - for line in lines: - ms = re.search(r':(/.+):(/.+)', line) - if ms: - home = ms.group(1) - shell = op.basename(ms.group(2)) - break - + try: + pwent = pwd.getpwnam(specialist) + home = pwent.pw_dir + shell = op.basename(pwent.pw_shell) + except KeyError: + pass if home != os.environ['HOME'] and op.exists(home): os.environ['HOME'] = home - return shell # # set environments for a specified specialist # def set_specialist_environments(specialist): - shell = set_specialist_home(specialist) resource = os.environ['HOME'] + "/.tcshrc" checkif = 0 # 0 outside of if; 1 start if, 2 check envs, -1 checked already missthen = 0 try: rf = open(resource, 'r') - except: + except OSError: return # skip if cannot open - nline = rf.readline() while nline: line = pgtrim(nline) @@ -1495,15 +1347,14 @@ def set_specialist_environments(specialist): missthen = 0 if re.match(r'^then$', line): continue # then on next line checkif = 0 # end of inline if - elif re.match(r'^endif', line): + elif line.startswith('endif'): checkif = 0 # end of if continue elif checkif == -1: # skip the line continue - elif checkif == 2 and re.match(r'^else', line): + elif checkif == 2 and line.startswith('else'): checkif = -1 # done check envs in if continue - if checkif == 1: if line == 'else': checkif = 2 @@ -1519,12 +1370,9 @@ def set_specialist_environments(specialist): if checkif == 1: continue else: continue - ms = re.match(r'^setenv\s+(.*)', line) if ms: one_specialist_environment(ms.group(1)) - rf.close() - SETPGLOG("HOMEBIN", PGLOG['PGBINDIR']) os.environ['PATH'] = add_local_path(PGLOG['HOMEBIN'], os.environ['PATH'], 0) @@ -1642,43 +1490,36 @@ def argv_to_string(argv = None, quote = 1, action = None): # convert an integer to non-10 based string # def int2base(x, base): - if x == 0: return '0' negative = 0 if x < 0: negative = 1 x = -x - dgts = [] while x: - dgts.append(str(int(x%base))) - x = int(x/base) + dgts.append(str(x % base)) + x //= base if negative: dgts.append('-') dgts.reverse() - return ''.join(dgts) # # convert a non-10 based string to an integer # def base2int(x, base): - if not isinstance(x, int): x = int(x) if x == 0: return 0 - negative = 0 if x < 0: negative = 1 x = -x - num = 0 fact = 1 while x: - num += (x%10)*fact + num += (x % 10) * fact fact *= base - x = int(x/10) + x //= 10 if negative: num = -num - return num # diff --git a/src/rda_python_common/PgLock.py b/src/rda_python_common/PgLock.py index 57d1cc0..ee4c3d8 100644 --- a/src/rda_python_common/PgLock.py +++ b/src/rda_python_common/PgLock.py @@ -419,12 +419,11 @@ def lock_host_update_control(cidx, pid, host, logact = 0): # # return lock information of a locked process # -def lock_process_info(pid, lockhost, runhost = None, pcnt = 0): - +def lock_process_info(pid, lockhost, runhost=None, pcnt=0): retstr = " {}<{}".format(lockhost, pid) if pcnt: retstr += "/{}".format(pcnt) retstr += ">" - if runhost and runhost != lockhost: retstr += '/' + runhost + if runhost is not None and runhost != lockhost: retstr += '/' + runhost return retstr # @@ -544,9 +543,8 @@ def lock_host_partition(pidx, pid, host, logact = 0): # update dsrqst lock info for given partition lock status # Return None if all is fine; error message otherwise # -def update_partition_lock(ridx, ptrec, logact = 0): - - if not ridx: return 0 +def update_partition_lock(ridx, ptrec, logact=0): + if not ridx: return None if logact: logerr = logact|PgLOG.ERRLOG logout = logact&(~PgLOG.EXITLG) @@ -558,27 +556,26 @@ def update_partition_lock(ridx, ptrec, logact = 0): cnd = "rindex = {}".format(ridx) pgrec = PgDBI.pgget(table, "pid, lockhost", cnd, logact|PgLOG.DOLOCK) if not pgrec: return "Error get Rqst{} record".format(ridx) # should not happen - if pgrec['pid'] > 0 and pgrec['lockhost'] != lockhost: - return "Rqst{} locked by non-lockhost process ({}/{})".format(ridx, pgrec['pid'], pgrec['lockhost']) - + return "Rqst{} locked by non-lockhost process ({}/{})".format(ridx, pgrec['pid'], pgrec['lockhost']) record = {} - if ptrec['pid'] > 0: + if ptrec.get('pid', 0) > 0: + # Locking a partition: increment the dsrqst pid counter. record['pid'] = pgrec['pid'] + 1 record['lockhost'] = lockhost record['locktime'] = ptrec['locktime'] else: + # Unlocking a partition: decrement the counter, but never below 0. if pgrec['pid'] > 1: pcnt = PgDBI.pgget('ptrqst', '', cnd + " AND pid > 0") if pgrec['pid'] > pcnt: pgrec['pid'] = pcnt - record['pid'] = pgrec['pid'] - 1 + record['pid'] = max(0, pgrec['pid'] - 1) record['lockhost'] = lockhost else: record['pid'] = 0 record['lockhost'] = '' if not PgDBI.pgupdt(table, record, cnd, logact): return "Error update Rqst{} lock".format(ridx) - return None # diff --git a/src/rda_python_common/PgOPT.py b/src/rda_python_common/PgOPT.py index 7b7e3cf..f21c2af 100644 --- a/src/rda_python_common/PgOPT.py +++ b/src/rda_python_common/PgOPT.py @@ -287,6 +287,10 @@ def get_default_info(opt): # # set output file name handler now # +# Kept in this module (instead of PgLOG, where the class version now lives) +# because applications reference PgOPT.OUTPUT directly; PgLOG.OUTPUT is kept in +# sync so PgLOG.pgexit() can close it. +# def open_output(outfile = None): global OUTPUT @@ -297,7 +301,9 @@ def open_output(outfile = None): except Exception as e: PgLOG.pglog("{}: Error open file to write - {}".format(outfile, str(e)), PGOPT['extlog']) else: # result to STDOUT + if OUTPUT and OUTPUT != sys.stdout: OUTPUT.close() OUTPUT = sys.stdout + PgLOG.OUTPUT = OUTPUT # # return 1 if valid infile names; sys.exit(1) otherwise @@ -316,15 +322,12 @@ def validate_infile_names(dsid): # validate an input filename against dsid # def validate_one_infile(infile, dsid): - ndsid = PgUtil.find_dataset_id(infile) - if ndsid == None: + if ndsid is None: return PgLOG.pglog("{}: No dsid identified in Input file name {}!".format(dsid, infile), PGOPT['extlog']) - fdsid = PgUtil.format_dataset_id(ndsid) if fdsid != dsid: return PgLOG.pglog("{}: Different dsid {} found in Input file name {}!".format(dsid, fdsid, infile), PGOPT['extlog']) - return PgLOG.SUCCESS # @@ -546,9 +549,7 @@ def build_record(flds, pgrec, tname, idx = 0): # set global variable PGOPT['UID'] with value of user.uid, fatal if unsuccessful # def set_uid(aname): - set_email_logact() - if 'LN' not in params: params['LN'] = PgLOG.PGLOG['CURUID'] elif params['LN'] != PgLOG.PGLOG['CURUID']: @@ -560,32 +561,25 @@ def set_uid(aname): msg = "'{}' runs '{} -{}' as '{}'!".format(PgLOG.PGLOG['CURUID'], aname, PGOPT['CACT'], params['LN']) PgLOG.pglog(msg, PGOPT['wrnlog']) PgLOG.set_specialist_environments(params['LN']) - if 'LN' not in params: PgLOG.pglog("Could not get user login name", PGOPT['extlog']) - validate_dataset() if OPTS[PGOPT['CACT']][2] > 0: validate_dsowner(aname) - - pgrec = PgDBI.pgget("dssdb.user", "uid", "logname = '{}' AND until_date IS NULL".format(params['LN']), PGOPT['extlog']) - if not pgrec: PgLOG.pglog("Could not get user.uid for " + params['LN'], PGOPT['extlog']) - PGOPT['UID'] = pgrec['uid'] - - open_output(params['OF'] if 'OF' in params else None) + uid = PgDBI.get_user_uid(params['LN']) + if not uid: PgLOG.pglog("Could not get user.uid for " + params['LN'], PGOPT['extlog']) + PGOPT['UID'] = uid + PgLOG.open_output(params['OF'] if 'OF' in params else None) # # set global variable PGOPT['UID'] as 0 for a sudo user # def set_sudo_uid(aname, uid): - set_email_logact() - if PgLOG.PGLOG['CURUID'] != uid: if 'DM' in params and re.match(r'^(start|begin)$', params['DM'], re.I): - msg = "'{}': must start Daemon '{} -{} as '{}'".format(PgLOG.PGLOG['CURUID'], aname, params['CACT'], uid) + msg = "'{}': must start Daemon '{} -{}' as '{}'".format(PgLOG.PGLOG['CURUID'], aname, PGOPT['CACT'], uid) else: - msg = "'{}': must run '{} -{}' as '{}'".format(PgLOG.PGLOG['CURUID'], aname, params['CACT'], uid) + msg = "'{}': must run '{} -{}' as '{}'".format(PgLOG.PGLOG['CURUID'], aname, PGOPT['CACT'], uid) PgLOG.pglog(msg, PGOPT['extlog']) - PGOPT['UID'] = 0 params['LN'] = PgLOG.PGLOG['CURUID'] @@ -593,16 +587,13 @@ def set_sudo_uid(aname, uid): # set global variable PGOPT['UID'] as 0 for root user # def set_root_uid(aname): - set_email_logact() - if PgLOG.PGLOG['CURUID'] != "root": if 'DM' in params and re.match(r'^(start|begin)$', params['DM'], re.I): - msg = "'{}': you must start Daemon '{} -{} as 'root'".format(PgLOG.PGLOG['CURUID'], aname, params['CACT']) + msg = "'{}': you must start Daemon '{} -{}' as 'root'".format(PgLOG.PGLOG['CURUID'], aname, PGOPT['CACT']) else: - msg = "'{}': you must run '{} -{}' as 'root'".format(PgLOG.PGLOG['CURUID'], aname, params['CACT']) + msg = "'{}': you must run '{} -{}' as 'root'".format(PgLOG.PGLOG['CURUID'], aname, PGOPT['CACT']) PgLOG.pglog(msg, PGOPT['extlog']) - PGOPT['UID'] = 0 params['LN'] = PgLOG.PGLOG['CURUID'] @@ -1034,22 +1025,19 @@ def get_option_key(p, flag = 0, skip = 0, lidx = 0, line = None, infile = None, # set values to given options, ignore options set in input files if the options # already set on command line # -def set_option_value(opt, val = None, cnl = 0, lidx = 0, line = None, infile = None): - +def set_option_value(opt, val=None, cnl=0, lidx=0, line=None, infile=None): if opt in CMDOPTS and lidx: # in input file, but given on command line already if opt not in params: params[opt] = CMDOPTS[opt] return - if val is None: val = '' if OPTS[opt][0]&3: if OPTS[opt][2]&16: if not val: val = 0 - elif re.match(r'^\d+$', val): + elif val.isdigit(): val = int(val) elif val and (opt == 'DS' or opt == 'OD'): val = PgUtil.format_dataset_id(val) - errmsg = None if not cnl and OPTS[opt][0]&3: if opt in params: @@ -1064,10 +1052,10 @@ def set_option_value(opt, val = None, cnl = 0, lidx = 0, line = None, infile = N ms = re.match(r'^!(\w*)', dstr) if ms: dstr = ms.group(1) - if vlen == 1 and dstr.find(val) > -1: errmsg = "{}: character must not be one of '{}'".format(val, str) + # Bug fix #3: was `str` (Python built-in), should be `dstr` + if vlen == 1 and dstr.find(val) > -1: errmsg = "{}: character must not be one of '{}'".format(val, dstr) elif vlen > 1 or (vlen == 0 and not OPTS[opt][2]&128) or (vlen == 1 and dstr.find(val) < 0): errmsg = "{} single-letter value must be one of '{}'".format(val, dstr) - if not errmsg: if OPTS[opt][0] == 2: # multiple value option if opt not in params: @@ -1085,7 +1073,7 @@ def set_option_value(opt, val = None, cnl = 0, lidx = 0, line = None, infile = N params[opt][rowidx] = val else: params[opt].append(val) # add next value - elif OPTS[opt][0] == 1: # single value option + elif OPTS[opt][0] == 1: # single value option if cnl and opt in params: if val: errmsg = "Multi-line value not allowed" elif OPTS[opt][2]&2 and PgUtil.pgcmp(params[opt], val): @@ -1105,13 +1093,11 @@ def set_option_value(opt, val = None, cnl = 0, lidx = 0, line = None, infile = N PGOPT['ACTS'] = OPTS[opt][0] # add action bit PGOPT['CACT'] = opt # add action name if opt == "SB": PGOPT['MSET'] = opt - if errmsg: if lidx: input_error(lidx, line, infile, "{}({}) - {}".format(opt, OPTS[opt][1], errmsg)) else: PgLOG.pglog("ERROR: {}({}) - {}".format(opt, OPTS[opt][1], errmsg), PGOPT['extlog']) - if not lidx: CMDOPTS[opt] = params[opt] # record options set on command lines # @@ -1450,7 +1436,6 @@ def set_file_format(count): # get frequency information # def get_control_frequency(frequency): - val = nf = 0 unit = None ms = re.match(r'^(\d+)([YMWDHNS])$', frequency, re.I) @@ -1464,20 +1449,19 @@ def get_control_frequency(frequency): nf = int(ms.group(2)) unit = 'M' if nf < 2 or nf > 10 or (30%nf): val = 0 - if not val: + # Bug fix #5: normalise all error paths to return (None, errmsg) if nf: - unit = "fraction of month frequency '{}' MUST be (2,3,5,6,10)".format(frequency) + errmsg = "fraction of month frequency '{}' MUST be (2,3,5,6,10)".format(frequency) elif unit: - val = "frequency '{}' MUST be larger than 0".format(frequency) + errmsg = "frequency '{}' MUST be larger than 0".format(frequency) elif re.search(r'/(\d+)$', frequency): - val = "fractional frequency '{}' for month ONLY".format(frequency) + errmsg = "fractional frequency '{}' for month ONLY".format(frequency) else: - val = "invalid frequency '{}', unit must be (Y,M,W,D,H)".format(frequency) - return (None, unit) - - freq = [0]*7 # initialize the frequence list - uidx = {'Y' : 0, 'D' : 2, 'H' : 3, 'N' : 4, 'S' : 5} + errmsg = "invalid frequency '{}', unit must be (Y,M,W,D,H,N,S)".format(frequency) + return (None, errmsg) + freq = [0]*7 # initialize the frequency list + uidx = {'Y': 0, 'D': 2, 'H': 3, 'N': 4, 'S': 5} if unit == 'M': freq[1] = val if nf: freq[6] = nf # number of fractions in a month @@ -1485,7 +1469,6 @@ def get_control_frequency(frequency): freq[2] = 7 * val elif unit in uidx: freq[uidx[unit]] = val - return (freq, unit) # @@ -1526,16 +1509,15 @@ def get_version_index(dsid, logact = 0): # # append given format (data or archive) sfmt to format string sformat # -def append_format_string(sformat, sfmt, chkend = 0): - +def append_format_string(sformat, sfmt, chkend=0): mp = r'(^|\.){}$' if chkend else r'(^|\.){}(\.|$)' if sfmt: if not sformat: sformat = sfmt else: for fmt in re.split(r'\.', sfmt): - if not re.search(mp.format(fmt), sformat, re.I): sformat += '.' + fmt - + # Bug fix #6: escape regex metacharacters in format component + if not re.search(mp.format(re.escape(fmt)), sformat, re.I): sformat += '.' + fmt return sformat # diff --git a/src/rda_python_common/PgSIG.py b/src/rda_python_common/PgSIG.py index 28ac6f0..44014c8 100644 --- a/src/rda_python_common/PgSIG.py +++ b/src/rda_python_common/PgSIG.py @@ -19,6 +19,8 @@ import errno import signal import time +import subprocess +import psutil from contextlib import contextmanager from . import PgLOG from . import PgDBI @@ -177,33 +179,51 @@ def stop_daemon(msg): PgLOG.PGLOG['LOGMASK'] |= PgLOG.MSGLOG # turn on logging before daemon stops PgLOG.pglog("{} Started at {}, Stopped gracefully{} by {}".format(PGSIG['DSTR'], PGSIG['STRTM'], msg, PgLOG.current_datetime()), PgLOG.LOGWRN) +# +# scan running processes via psutil and return a list of dicts matching aname/uname +# +# Mirrors the semantics of the previous "ps -u U -f | grep A" / "ps -C A -f" pipelines: +# - with uname: substring match of aname anywhere in cmdline (looser, like grep) +# - without uname: aname matches the executable basename (with optional .ext) +# +# Each entry has keys: 'pid', 'ppid', 'args' ('args' is cmdline[1:] joined). +# +def _scan_app_processes(aname, uname=None): + results = [] + for proc in psutil.process_iter(['pid', 'ppid', 'username', 'cmdline']): + try: + info = proc.info + cmdline = info.get('cmdline') or [] + if not cmdline: continue + if uname is not None and info.get('username') != uname: continue + if uname: + if not any(aname in arg for arg in cmdline): continue + else: + exe = os.path.basename(cmdline[0]) + if exe != aname and re.sub(r'\.\w+$', '', exe) != aname: continue + results.append({ + 'pid': info['pid'], + 'ppid': info['ppid'], + 'args': ' '.join(cmdline[1:]), + }) + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + return results + # # check if a daemon is running already # # aname - application name for the daemon # uname - user login name who started the daemon -# +# # return the process id if yes and 0 if not # -def check_daemon(aname, uname = None): - - if uname: - check_vuser(uname, aname) - pcmd = "ps -u {} -f | grep {} | grep ' 1 '".format(uname, aname) - mp = r"^\s*{}\s+(\d+)\s+1\s+".format(uname) - else: - pcmd = "ps -C {} -f | grep ' 1 '".format(aname) - mp = r"^\s*\w+\s+(\d+)\s+1\s+" - - buf = PgLOG.pgsystem(pcmd, PgLOG.LOGWRN, 20+1024) - if buf: - cpid = os.getpid() - lines = buf.split('\n') - for line in lines: - ms = re.match(mp, line) - pid = int(ms.group(1)) if ms else 0 - if pid > 0 and pid != cpid: return pid - +def check_daemon(aname, uname=None): + if uname: check_vuser(uname, aname) + cpid = os.getpid() + for p in _scan_app_processes(aname, uname): + if p['ppid'] != 1: continue + if p['pid'] != cpid: return p['pid'] return 0 # @@ -215,39 +235,24 @@ def check_daemon(aname, uname = None): # # return the process id if yes and 0 if not # -def check_application(aname, uname = None, sargv = None): - - if uname: - check_vuser(uname, aname) - pcmd = "ps -u {} -f | grep {} | grep -v ' grep '".format(uname, aname) - mp = r"^\s*{}\s+(\d+)\s+(\d+)\s+.*{}\S*\s+(.*)$".format(uname, aname) - else: - pcmd = "ps -C {} -f".format(aname) - mp = r"^\s*\w+\s+(\d+)\s+(\d+)\s+.*{}\S*\s+(.*)$".format(aname) - - buf = PgLOG.pgsystem(pcmd, PgLOG.LOGWRN, 20+1024) - if not buf: return 0 - +def check_application(aname, uname=None, sargv=None): + if uname: check_vuser(uname, aname) + procs = _scan_app_processes(aname, uname) + if not procs: return 0 cpids = [os.getpid(), os.getppid()] pids = [] ppids = [] astrs = [] - lines = buf.split('\n') - for line in lines: - ms = re.match(mp, line) - if not ms: continue - pid = int(ms.group(1)) - ppid = int(ms.group(2)) + for p in procs: + pid, ppid = p['pid'], p['ppid'] if pid in cpids: if ppid not in cpids: cpids.append(ppid) continue pids.append(pid) ppids.append(ppid) - if sargv: astrs.append(ms.group(3)) - + if sargv: astrs.append(p['args']) pcnt = len(pids) if not pcnt: return 0 - i = 0 while i < pcnt: pid = pids[i] @@ -258,26 +263,24 @@ def check_application(aname, uname = None, sargv = None): i = 0 else: i += 1 - for i in range(pcnt): pid = pids[i] if pid and (not sargv or sargv.find(astrs[i]) > -1): return pid - return 0 # # validate if the current process is a single one. Quit if not # -def validate_single_process(aname, uname = None, sargv = None, logact = PgLOG.LOGWRN): - - pid = check_application(aname, uname, sargv) - if pid: - msg = aname - if sargv: msg += ' ' + sargv - msg += ": already running as PID={} on {}".format(pid, PgLOG.PGLOG['HOSTNAME']) - if uname: msg += ' By ' + uname - PgLOG.pglog(msg + ', Quit Now', logact) - sys.exit(0) +def validate_single_process(aname, uname=None, sargv=None, logact=None): + if logact is None: logact = PgLOG.LOGWRN + pid = check_application(aname, uname, sargv) + if pid: + msg = aname + if sargv: msg += ' ' + sargv + msg += ": already running as PID={} on {}".format(pid, PgLOG.PGLOG['HOSTNAME']) + if uname: msg += ' By ' + uname + PgLOG.pglog(msg + ', Quit Now', logact) + sys.exit(0) # # check how many processes are running for an application already @@ -288,29 +291,16 @@ def validate_single_process(aname, uname = None, sargv = None, logact = PgLOG.LO # # return the the number of processes (exclude the child one) # -def check_multiple_application(aname, uname = None, sargv = None): - - if uname: - check_vuser(uname, aname) - pcmd = "ps -u {} -f | grep {} | grep -v ' grep '".format(uname, aname) - mp = r"^\s*{}\s+(\d+)\s+(\d+)\s+.*{}\S*\s+(.*)$".format(uname, aname) - else: - pcmd = "ps -C {} -f".format(aname) - mp = r"^\s*\w+\s+(\d+)\s+(\d+)\s+.*{}\S*\s+(.*)$".format(aname) - - buf = PgLOG.pgsystem(pcmd, PgLOG.LOGWRN, 20+1024) - if not buf: return 0 - +def check_multiple_application(aname, uname=None, sargv=None): + if uname: check_vuser(uname, aname) + procs = _scan_app_processes(aname, uname) + if not procs: return 0 dpids = [os.getpid(), os.getppid()] pids = [] ppids = [] astrs = [] - lines = buf.split('\n') - for line in lines: - ms = re.match(mp, line) - if not ms: continue - pid = int(ms.group(1)) - ppid = int(ms.group(2)) + for p in procs: + pid, ppid = p['pid'], p['ppid'] if pid in dpids: if ppid > 1 and ppid not in dpids: dpids.append(ppid) continue @@ -319,11 +309,9 @@ def check_multiple_application(aname, uname = None, sargv = None): continue pids.append(pid) ppids.append(ppid) - if sargv: astrs.append(ms.group(3)) - + if sargv: astrs.append(p['args']) pcnt = len(pids) if not pcnt: return 0 - i = 0 while i < pcnt: pid = pids[i] @@ -338,26 +326,24 @@ def check_multiple_application(aname, uname = None, sargv = None): i = pids[i] = 0 continue i += 1 - ccnt = 0 for i in range(pcnt): if pids[i] and (not sargv or sargv.find(astrs[i]) > -1): ccnt += 1 - return ccnt # # validate if the running processes reach the limit for the given app; Quit if yes # -def validate_multiple_process(aname, plimit, uname = None, sargv = None, logact = PgLOG.LOGWRN): - - pcnt = check_multiple_application(aname, uname, sargv) - if pcnt >= plimit: - msg = aname - if sargv: msg += ' ' + sargv - msg += ": already running in {} processes on {}".format(pcnt, PgLOG.PGLOG['HOSTNAME']) - if uname: msg += ' By ' + uname - PgLOG.pglog(msg + ', Quit Now', logact) - sys.exit(0) +def validate_multiple_process(aname, plimit, uname=None, sargv=None, logact=None): + if logact is None: logact = PgLOG.LOGWRN + pcnt = check_multiple_application(aname, uname, sargv) + if pcnt >= plimit: + msg = aname + if sargv: msg += ' ' + sargv + msg += ": already running in {} processes on {}".format(pcnt, PgLOG.PGLOG['HOSTNAME']) + if uname: msg += ' By ' + uname + PgLOG.pglog(msg + ', Quit Now', logact) + sys.exit(0) # # fork process @@ -365,18 +351,16 @@ def validate_multiple_process(aname, plimit, uname = None, sargv = None, logact # return the defined result from call of fork # def process_fork(dstr): - for i in range(10): # try 10 times try: pid = os.fork() return pid except OSError as e: - if e.errno == errno.EAGAIN: - os.sleep(5) + if e.errno == errno.EAGAIN: + time.sleep(5) # Bug fix: os.sleep() does not exist; use time.sleep() else: PgLOG.pglog("{}: {}".format(dstr, str(e)), PgLOG.LGEREX) break - PgLOG.pglog("{}: too many tries (10) for os.fork()".format(dstr), PgLOG.LGEREX) # @@ -451,16 +435,14 @@ def kill_process(pid, signum, logact = 0): # wait child process to finish # def clean_dead_child(signum, frame): - live = 0 - while True: try: dpid, status = os.waitpid(-1, os.WNOHANG) except ChildProcessError as e: break # no child process any more except Exception as e: - PgLOG.PGLOG("Error check child process: {}".format(str(e)), PgLOG.ERRLOG) + PgLOG.pglog("Error check child process: {}".format(str(e)), PgLOG.ERRLOG) break else: if dpid == 0: @@ -502,10 +484,9 @@ def signal_daemon(sname, aname, uname): # # start a time child to run the command in case hanging # -def timeout_command(cmd, logact = PgLOG.LOGWRN, cmdopt = 4): - +def timeout_command(cmd, logact=None, cmdopt=4): + if logact is None: logact = PgLOG.LOGWRN if logact&PgLOG.EXITLG: logact &= ~PgLOG.EXITLG - PgLOG.pglog("> " + cmd, logact) if start_timeout_child(cmd, logact): PgLOG.pgsystem(cmd, logact, cmdopt) @@ -516,10 +497,9 @@ def timeout_command(cmd, logact = PgLOG.LOGWRN, cmdopt = 4): # # return: 1 - in child, 0 - in parent # -def start_timeout_child(msg, logact = PgLOG.LOGWRN): - +def start_timeout_child(msg, logact=None): + if logact is None: logact = PgLOG.LOGWRN pid = process_fork(msg) - if pid == 0: # in child signal.signal(signal.SIGQUIT, signal_catch) # catch quit signal only PGSIG['PPID'] = PGSIG['PID'] @@ -527,57 +507,51 @@ def start_timeout_child(msg, logact = PgLOG.LOGWRN): PgLOG.cmdlog("Timeout child to " + msg, time.time(), 0) PgDBI.pgdisconnect(0) # disconnect database in child return 1 - # in parent for i in range(PgLOG.PGLOG['TIMEOUT']): if not check_process(pid): break - sys.sleep(2) - - if check_process(pid): + time.sleep(2) # Bug fix: sys.sleep() does not exist; use time.sleep() + if check_process(pid): # Bug fix: removed extra 'self' argument msg += ": timeout({} secs) in CPID {}".format(2*PgLOG.PGLOG['TIMEOUT'], pid) pids = kill_children(pid, 0) - sys.sleep(6) + time.sleep(6) # Bug fix: sys.sleep() does not exist; use time.sleep() if kill_process(pid, signal.SIGKILL, PgLOG.LOGERR): pids.insert(0, pid) - if pids: msg += "\nProcess({}) Killed".format(','.join(map(str, pids))) PgLOG.pglog(msg, logact) - return 0 # # kill children recursively start from the deepest and return the pids got killed # -def kill_children(pid, logact = PgLOG.LOGWRN): - - buf = PgLOG.pgsystem("ps --ppid {} -o pid".format(pid), logact, 20) +def kill_children(pid, logact=None): + if logact is None: logact = PgLOG.LOGWRN pids = [] - if buf: - lines = buf.split('\n') - for line in lines: - ms = re.match(r'^\s*(\d+)', line) - if not ms: continue - cid = int(ms.group(1)) - if not check_process(cid): continue - cids = kill_children(cid, logact) - if cids: pids = cids + pids - if kill_process(cid, signal.SIGKILL, logact) == PgLOG.SUCCESS: pids.insert(0, cid) - + try: + children = psutil.Process(pid).children() + except psutil.NoSuchProcess: + children = [] + except Exception as e: + PgLOG.pglog("Error listing children of pid {}: {}".format(pid, str(e)), logact) + children = [] + for child in children: + cid = child.pid + if not check_process(cid): continue + cids = kill_children(cid, logact) + if cids: pids = cids + pids + if kill_process(cid, signal.SIGKILL, logact) == PgLOG.SUCCESS: pids.insert(0, cid) if logact and len(pids): PgLOG.pglog("Process({}) Killed".format(','.join(map(str, pids))), logact) - return pids # # start a child process # pname - unique process name # -def start_child(pname, logact = PgLOG.LOGWRN, dowait = 0): - +def start_child(pname, logact=None, dowait=0): global CBIDS + if logact is None: logact = PgLOG.LOGWRN if PGSIG['MPROC'] < 2: return 1 # no need child process - if logact&PgLOG.EXITLG: logact &= ~PgLOG.EXITLG if logact&PgLOG.MSGLOG: logact |= PgLOG.FRCLOG - if PGSIG['QUIT']: return PgLOG.pglog("{} is in QUIT mode, cannot start CPID for {}".format(PGSIG['DSTR'], pname), logact) elif len(CPIDS) >= PGSIG['MPROC']: @@ -590,9 +564,7 @@ def start_child(pname, logact = PgLOG.LOGWRN, dowait = 0): i += 1 else: return PgLOG.pglog("{}-{}: {} child processes already running at {}".format(PGSIG['DSTR'], pname, pcnt, PgLOG.current_datetime()), logact) - if check_child(pname): return -1 # process is running already - pid = process_fork(PGSIG['DSTR']) if pid: CPIDS[pid] = pname # record the child process id @@ -607,7 +579,6 @@ def start_child(pname, logact = PgLOG.LOGWRN, dowait = 0): PGSIG['DSTR'] += ": CPID {} for {}".format(pid, pname) PgLOG.cmdlog("CPID {} for {}".format(pid, pname)) PgDBI.pgdisconnect(0) # disconnect database in child - return 1 # child started successfully # @@ -628,10 +599,9 @@ def pname2cpid(pname): # return the number of running processes if dowait == 0 or 1 # return the number of none-running processes if dowait == -1 # -def check_child(pname, pid = 0, logact = PgLOG.LOGWRN, dowait = 0): - +def check_child(pname, pid=0, logact=None, dowait=0): + if logact is None: logact = PgLOG.LOGWRN if PGSIG['MPROC'] < 2: return 0 # no child process - if logact&PgLOG.EXITLG: logact &= ~PgLOG.EXITLG ccnt = i = 0 if dowait < 0: ccnt = 1 if (pid or pname) else PGSIG['MPROC'] @@ -654,11 +624,9 @@ def check_child(pname, pid = 0, logact = PgLOG.LOGWRN, dowait = 0): pcnt += 1 elif cpid in CPIDS: del CPIDS[cpid] - if pcnt == 0 or dowait == 0 or pcnt < ccnt: break show_wait_message(i, "{}: wait {}/{} child processes".format(PGSIG['DSTR'], pcnt, PGSIG['MPROC']), logact, dowait) i += 1 - return (ccnt - pcnt) if ccnt else pcnt # @@ -697,21 +665,17 @@ def start_none_daemon(aname, cact = None, uname = None, mproc = 1, wtime = 120, # pmsg - process message if given # def check_process(pid): - - buf = PgLOG.pgsystem("ps -p {} -o pid".format(pid), PgLOG.LGWNEX, 20) - if buf: - mp = r'^\s*{}$'.format(pid) - lines = buf.split('\n') - for line in lines: - if re.match(mp, line): return 1 - - return 0 + try: + os.kill(pid, 0) + except OSError: + return 0 + return 1 # # check a process id on give host # -def check_host_pid(host, pid, pmsg = None, logact = PgLOG.LOGWRN): - +def check_host_pid(host, pid, pmsg=None, logact=None): + if logact is None: logact = PgLOG.LOGWRN cmd = 'rdaps' if host: cmd += " -h " + host cmd += " -p {}".format(pid) @@ -731,8 +695,8 @@ def check_host_pid(host, pid, pmsg = None, logact = PgLOG.LOGWRN): # # return 1 if process is steal live, 0 died already, -1 error checking # -def check_host_process(host, pid, ppid = 0, uname = None, aname = None, pmsg = None, logact = PgLOG.LOGWRN): - +def check_host_process(host, pid, ppid=0, uname=None, aname=None, pmsg=None, logact=None): + if logact is None: logact = PgLOG.LOGWRN cmd = "rdaps" if host: cmd += " -h " + host if pid: cmd += " -p {}".format(pid) @@ -744,44 +708,10 @@ def check_host_process(host, pid, ppid = 0, uname = None, aname = None, pmsg = N if pmsg: PgLOG.pglog(pmsg, logact&(~PgLOG.EXITLG)) return 1 -# -# get a single slurm status record -# -def get_slurm_info(bcmd, logact = PgLOG.LOGWRN): - - stat = {} - buf = PgLOG.pgsystem(bcmd, logact, 16) - if not buf: return stat - - chkt = 1 - lines = buf.split('\n') - for line in lines: - if chkt: - if re.match(r'^\s*JOBID\s', line, re.I): - ckeys = re.split(r'\s+', PgLOG.pgtrim(line)) - kcnt = len(ckeys) - chkt = 0 - else: - if re.match(r'^-----', line): continue - vals = re.split(r'\s+', PgLOG.pgtrim(line)) - vcnt = len(vals) - if vcnt >= kcnt: - for i in range(kcnt): - ckeys[i] = ckeys[i].upper() - stat[ckeys[i]] = vals[i] - - if vcnt > kcnt: - for i in range(kcnt, vcnt): - stat[ckeys[kcnt-1]] += ' ' + str(vals[i]) - break - - return stat - # # get a single pbs status record via qstat # -def get_pbs_info(qopts, multiple = 0, logact = 0, chkcnt = 1): - +def get_pbs_info(qopts, multiple=0, logact=0, chkcnt=1): stat = {} loop = 0 buf = None @@ -790,9 +720,7 @@ def get_pbs_info(qopts, multiple = 0, logact = 0, chkcnt = 1): if buf: break loop += 1 time.sleep(6) - if not buf: return stat - chkt = chkd = 1 lines = buf.split('\n') for line in lines: @@ -803,7 +731,7 @@ def get_pbs_info(qopts, multiple = 0, logact = 0, chkcnt = 1): ckeys[1] = 'UserName' ckeys[3] = 'JobName' ckeys[7] = 'Reqd' + ckeys[7] - ckeys[8] = 'Reqd' + ckeys[7] + ckeys[8] = 'Reqd' + ckeys[8] # Bug fix: was ckeys[7] (wrong index) ckeys[9] = 'State' ckeys[10] = 'Elap' + ckeys[7] ckeys.append('Node') @@ -832,73 +760,24 @@ def get_pbs_info(qopts, multiple = 0, logact = 0, chkcnt = 1): else: stat[ckeys[i]] = vals[i] if vcnt == kcnt: break - return stat -# -# get multiple slurn status record -# -def get_slurm_multiple(bcmd, logact = PgLOG.LOGWRN): - - buf = PgLOG.pgsystem(bcmd, logact, 16) - if not buf: return 0 - - stat = {} - j = 0 - chkt = chkd = 1 - lines = buf.split('\n') - for line in lines: - if chkt: - if re.match(r'^\s*JOBID\s', line, re.I): - ckeys = re.split(r'\s+', PgLOG.pgtrim(line)) - kcnt = len(ckeys) - for i in range(kcnt): - ckeys[i] = ckeys[i].upper() - stat[ckeys[i]] = [] - chkt = 0 - elif chkd: - if re.match(r'^-----', line): chkd = 0 - else: - vals = re.split(r'\s+', PgLOG.pgtrim(line)) - vcnt = len(vals) - if vcnt >= kcnt: - for i in range(kcnt): - stat[ckeys[i]].append(vals[i]) - - if vcnt > kcnt: - for i in range(kcnt, vcnt): - stat[ckeys[kcnt-1]][j] += ' ' + str(vals[i]) - j += 1 - - return stat if j else 0 - -# -# check status of a slurm batch id -# bid - specified batch id -# -# return hash of batch status, 0 if cannot check any more -# -def check_slurm_status(bid, logact = PgLOG.LOGWRN): - - return get_slurm_info("sacct -o jobid,user,totalcpu,elapsed,ncpus,state,jobname,nodelist -j {}".format(bid), logact) - # # check status of a pbs batch id # bid - specified batch id # # return hash of batch status, 0 if cannot check any more # -def check_pbs_status(bid, logact = PgLOG.LOGWRN): - +def check_pbs_status(bid, logact=None): + if logact is None: logact = PgLOG.LOGWRN stat = {} buf = PgLOG.pgsystem("qhist -w -j {}".format(bid), logact, 20) if not buf: return stat - chkt = 1 lines = buf.split('\n') for line in lines: if chkt: - if re.match(r'^Job', line): + if line.startswith('Job'): line = re.sub(r'^Job ID', 'JobID', line, 1) line = re.sub(r'Finish Time', 'FinishTime', line, 1) line = re.sub(r'Req Mem', 'ReqMem', line, 1) @@ -915,39 +794,17 @@ def check_pbs_status(bid, logact = PgLOG.LOGWRN): for i in range(kcnt): stat[ckeys[i]] = vals[i] break - return stat -# -# check if a slurm batch id is live -# bid - specified batch id -# -# return 1 if process is steal live, 0 died already or error checking -# -def check_slurm_process(bid, pmsg = None, logact = PgLOG.LOGWRN): - - stat = get_slurm_info("squeue -l -j {}".format(bid), logact) - - if stat: - ms = re.match(r'^(RUNNING|PENDING|SUSPENDE|COMPLETI|CONFIGUR|REQUEUE_)$', stat['STATE']) - if ms: - if pmsg: PgLOG.pglog("{}, STATE={}".format(pmsg, ms.group(1)), logact&~PgLOG.EXITLG) - return 1 - else: - return 0 - - return -1 - # # check if a pbs batch id is live # bid - specified batch id # # return 1 if process is steal live, 0 died already or error checking # -def check_pbs_process(bid, pmsg = None, logact = PgLOG.LOGWRN): - +def check_pbs_process(bid, pmsg=None, logact=None): + if logact is None: logact = PgLOG.LOGWRN stat = get_pbs_info(bid, 0, logact) - ret = -1 if stat: ms = re.match(r'^(B|R|Q|S|H|W|X)$', stat['State']) @@ -959,48 +816,40 @@ def check_pbs_process(bid, pmsg = None, logact = PgLOG.LOGWRN): ret = 0 elif pmsg: pmsg += ", Process Not Exists and returns -1" - if pmsg: PgLOG.pglog(pmsg, logact&~PgLOG.EXITLG) - return ret # # get wait time # def get_wait_time(wtime, default, tmsg): - if not wtime: wtime = default # use default time - if type(wtime) is int: return wtime if re.match(r'^(\d*)$', wtime): return int(wtime) - ms = re.match(r'^(\d*)([DHMS])$', wtime, re.I) if ms: ret = int(ms.group(1)) unit = ms.group(2) else: PgLOG.pglog("{}: '{}' NOT in (D,H,M,S)".format(wtime, tmsg), PgLOG.LGEREX) - + return default # Bug fix: LGEREX exits, but add fallback so 'unit' is always defined if unit != 'S': ret *= 60 # seconds in a minute if unit != 'M': ret *= 60 # minutes in an hour if unit != 'H': ret *= 24 # hours in a day - return ret # in seconds # # start a background process and record its id; check PgLOG.pgsystem() in PgLOG.pm for # valid cmdopt values # -def start_background(cmd, logact = PgLOG.LOGWRN, cmdopt = 5, dowait = 0): - +def start_background(cmd, logact=None, cmdopt=5, dowait=0): + if logact is None: logact = PgLOG.LOGWRN if PGSIG['BPROC'] < 2: return PgLOG.pgsystem(cmd, logact, cmdopt) # no background - act = logact&(~PgLOG.EXITLG) if act&PgLOG.MSGLOG: act |= PgLOG.FRCLOG # make sure background calls always logged - if len(CBIDS) >= PGSIG['BPROC']: i = 0 while True: @@ -1011,7 +860,6 @@ def start_background(cmd, logact = PgLOG.LOGWRN, cmdopt = 5, dowait = 0): i += 1 else: return PgLOG.pglog("{}-{}: {} background calls already at {}".format(PGSIG['DSTR'], cmd, bcnt, PgLOG.current_datetime()), act) - cmdlog = (act if cmdopt&1 else PgLOG.WARNLG) if cmdopt&8: PgLOG.cmdlog("starts '{}'".format(cmd), None, cmdlog) @@ -1020,14 +868,15 @@ def start_background(cmd, logact = PgLOG.LOGWRN, cmdopt = 5, dowait = 0): bckcmd = cmd if cmdopt&2: bckcmd += " >> {}/{}".format(PgLOG.PGLOG['LOGPATH'], PgLOG.PGLOG['LOGFILE']) - if cmdopt&4: if not PgLOG.PGLOG['ERRFILE']: PgLOG.PGLOG['ERRFILE'] = re.sub(r'\.log$', '.err', PgLOG.PGLOG['LOGFILE'], 1) bckcmd += " 2>> {}/{}".format(PgLOG.PGLOG['LOGPATH'], PgLOG.PGLOG['ERRFILE']) - bckcmd += " &" - os.system(bckcmd) + # shell=True is required for the redirections (>> / 2>>) and trailing '&'; + # the '&' makes the shell fork the command and exit, so the command gets + # reparented to init (ppid=1), matching the lookup logic in record_background(). + subprocess.Popen(bckcmd, shell=True) return record_background(cmd, logact) # @@ -1045,10 +894,9 @@ def bcmd2cbid(bcmd): # bid - check this specified background process id if given # return the number of processes are still running # -def check_background(bcmd, bid = 0, logact = PgLOG.LOGWRN, dowait = 0): - +def check_background(bcmd, bid=0, logact=None, dowait=0): + if logact is None: logact = PgLOG.LOGWRN if PGSIG['BPROC'] < 2: return 0 # no background process - if logact&PgLOG.EXITLG: logact &= ~PgLOG.EXITLG if not bid and bcmd: bid = bcmd2cbid(bcmd) bcnt = i = 0 @@ -1063,50 +911,53 @@ def check_background(bcmd, bid = 0, logact = PgLOG.LOGWRN, dowait = 0): elif bid in CBIDS: del CBIDS[bid] # clean the saved info for the process elif not bcmd: - for bid in CBIDS: + # Bug fix: cannot delete from a dict while iterating over it; + # iterate over a copy of the keys instead. + cbids = list(CBIDS) + for bid in cbids: if check_process(bid): # process is not done yet bcnt += 1 else: del CBIDS[bid] - if not (bcnt and dowait): break show_wait_message(i, "{}: wait {}/{} background processes".format(PGSIG['DSTR'], bcnt, PGSIG['MPROC']), logact, dowait) i += 1 bcnt = 0 - return bcnt # # check and record process id for background command; return 1 if success full; # 0 otherwise; -1 if done already # -def record_background(bcmd, logact = PgLOG.LOGWRN): - +def record_background(bcmd, logact=None): + if logact is None: logact = PgLOG.LOGWRN ms = re.match(r'^(\S+)', bcmd) - if ms: - aname = ms.group(1) - else: - aname = bcmd - - mp = r"^\s*(\S+)\s+(\d+)\s+1\s+.*{}(.*)$".format(aname) - pc = "ps -u {},{} -f | grep ' 1 ' | grep {}".format(PgLOG.PGLOG['CURUID'], PgLOG.PGLOG['COMMONUSER'], aname) + aname = ms.group(1) if ms else bcmd + curuid = PgLOG.PGLOG['CURUID'] + commonuser = PgLOG.PGLOG['COMMONUSER'] for i in range(2): - buf = PgLOG.pgsystem(pc, logact, 20+1024) - if buf: - lines = buf.split('\n') - for line in lines: - ms = re.match(mp, line) - if not ms: continue - (uid, sbid, acmd) = ms.groups() - bid = int(sbid) + for proc in psutil.process_iter(['pid', 'ppid', 'username', 'cmdline']): + try: + info = proc.info + if info.get('ppid') != 1: continue + uid = info.get('username') + if uid != curuid and uid != commonuser: continue + cmdline = info.get('cmdline') or [] + if not cmdline: continue + line = ' '.join(cmdline) + idx = line.find(aname) + if idx < 0: continue + bid = info['pid'] if bid in CBIDS: return -1 - if uid == PgLOG.PGLOG['COMMONUSER']: + acmd = line[idx+len(aname):] + if uid == commonuser: acmd = re.sub(r'^\.(pl|py)\s+', '', acmd, 1) if re.match(r'^{}{}'.format(aname, acmd), bcmd): continue CBIDS[bid] = bcmd return 1 + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue time.sleep(2) - return 0 # @@ -1129,11 +980,10 @@ def sleep_daemon(wtime = 0, mtime = None): # # show wait message every dintv and then sleep for PGSIG['WTIME'] # -def show_wait_message(loop, msg, logact = PgLOG.LOGWRN, dowait = 0): - +def show_wait_message(loop, msg, logact=None, dowait=0): + if logact is None: logact = PgLOG.LOGWRN if loop > 0 and (loop%30) == 0: PgLOG.pglog("{} at {}".format(msg, PgLOG.current_datetime()), logact) - if dowait: time.sleep(PGSIG['WTIME']) # @@ -1156,9 +1006,7 @@ def raise_pgtimeout(signum, frame): raise TimeoutError def timeout_func(): - # Add a timeout block. - with pgtimeout(1): - print('entering block') - import time - time.sleep(10) - print('This should never get printed because the line before timed out') + with pgtimeout(1): + print('entering block') + time.sleep(10) + print('This should never get printed because the line before timed out') diff --git a/src/rda_python_common/PgSplit.py b/src/rda_python_common/PgSplit.py index 906f823..61b2c33 100644 --- a/src/rda_python_common/PgSplit.py +++ b/src/rda_python_common/PgSplit.py @@ -24,10 +24,9 @@ # and return the records need to be added, modified and deleted # def compare_wfile(wfrecs, dsrecs): - - flds = dsrecs.keys() + flds = list(dsrecs.keys()) flen = len(flds) - arecs = dict(zip(flds, [[]]*flen)) + arecs = {fld: [] for fld in flds} mrecs = {} drecs = [] wfcnt = len(wfrecs['wid']) @@ -60,9 +59,7 @@ def compare_wfile(wfrecs, dsrecs): arecs[fld].extend(wfrecs[fld][i:wfcnt]) elif j < dscnt: drecs.extend(dsrecs['wid'][j:dscnt]) - if len(arecs['wid']) == 0: arecs = {} - return (arecs, mrecs, drecs) # @@ -118,16 +115,14 @@ def get_dsid_condition(dsid, condition): # # insert one record into wfile and/or wfile_dsid # -def pgadd_wfile(dsid, wfrec, logact = PgLOG.LOGERR, getid = None): - - - record = {'wfile' : wfrec['wfile'], - 'dsid' : (wfrec['dsid'] if 'dsid' in wfrec else dsid)} +def pgadd_wfile(dsid, wfrec, logact=None, getid=None): + if logact is None: logact = PgLOG.LOGERR + record = {'wfile': wfrec['wfile'], + 'dsid': (wfrec['dsid'] if 'dsid' in wfrec else dsid)} wret = PgDBI.pgadd('wfile', record, logact, 'wid') if wret: record = wfile2wdsid(wfrec, wret) PgDBI.pgadd('wfile_' + dsid, record, logact|PgLOG.ADDTBL) - if logact&PgLOG.AUTOID or getid: return wret else: @@ -136,16 +131,15 @@ def pgadd_wfile(dsid, wfrec, logact = PgLOG.LOGERR, getid = None): # # insert multiple records into wfile and/or wfile_dsid # -def pgmadd_wfile(dsid, wfrecs, logact = PgLOG.LOGERR, getid = None): - - records = {'wfile' : wfrecs['wfile'], - 'dsid' : (wfrecs['dsid'] if 'dsid' in wfrecs else [dsid]*len(wfrecs['wfile']))} +def pgmadd_wfile(dsid, wfrecs, logact=None, getid=None): + if logact is None: logact = PgLOG.LOGERR + records = {'wfile': wfrecs['wfile'], + 'dsid': (wfrecs['dsid'] if 'dsid' in wfrecs else [dsid]*len(wfrecs['wfile']))} wret = PgDBI.pgmadd('wfile', records, logact, 'wid') wcnt = wret if isinstance(wret, int) else len(wret) if wcnt: records = wfile2wdsid(wfrecs, wret) PgDBI.pgmadd('wfile_' + dsid, records, logact|PgLOG.ADDTBL) - if logact&PgLOG.AUTOID or getid: return wret else: @@ -155,8 +149,8 @@ def pgmadd_wfile(dsid, wfrecs, logact = PgLOG.LOGERR, getid = None): # update one or multiple rows in wfile and/or wfile_dsid # exclude dsid in condition # -def pgupdt_wfile(dsid, wfrec, condition, logact = PgLOG.LOGERR): - +def pgupdt_wfile(dsid, wfrec, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR record = trim_wfile_fields(wfrec) if record: wret = PgDBI.pgupdt('wfile', record, get_dsid_condition(dsid, condition), logact) @@ -165,15 +159,14 @@ def pgupdt_wfile(dsid, wfrec, condition, logact = PgLOG.LOGERR): if wret: record = wfile2wdsid(wfrec) if record: wret = PgDBI.pgupdt("wfile_" + dsid, record, condition, logact|PgLOG.ADDTBL) - return wret # # update one row in wfile and/or wfile_dsid with dsid change # exclude dsid in condition # -def pgupdt_wfile_dsid(dsid, odsid, wfrec, wid, logact = PgLOG.LOGERR): - +def pgupdt_wfile_dsid(dsid, odsid, wfrec, wid, logact=None): + if logact is None: logact = PgLOG.LOGERR record = trim_wfile_fields(wfrec) cnd = 'wid = {}'.format(wid) if record: @@ -195,39 +188,36 @@ def pgupdt_wfile_dsid(dsid, odsid, wfrec, wid, logact = PgLOG.LOGERR): doupdt = False if doupdt and record: wret = PgDBI.pgupdt(tname, record, cnd, logact|PgLOG.ADDTBL) - return wret # # delete one or multiple rows in wfile and/or wfile_dsid, and add the record(s) into wfile_delete # exclude dsid in conidtion # -def pgdel_wfile(dsid, condition, logact = PgLOG.LOGERR): - +def pgdel_wfile(dsid, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR pgrecs = pgmget_wfile(dsid, '*', condition, logact|PgLOG.ADDTBL) - wret = PgDBI.pgdel('wfile', get_dsid_condition(dsid, condition), logact) + wret = PgDBI.pgdel('wfile', get_dsid_condition(dsid, condition), logact) if wret: PgDBI.pgdel("wfile_" + dsid, condition, logact) if wret and pgrecs: PgDBI.pgmadd('wfile_delete', pgrecs, logact) - return wret # # delete one or multiple rows in sfile, and add the record(s) into sfile_delete # -def pgdel_sfile(condition, logact = PgLOG.LOGERR): - +def pgdel_sfile(condition, logact=None): + if logact is None: logact = PgLOG.LOGERR pgrecs = PgDBI.pgmget('sfile', '*', condition, logact) - sret = PgDBI.pgdel('sfile', condition, logact) + sret = PgDBI.pgdel('sfile', condition, logact) if sret and pgrecs: PgDBI.pgmadd('sfile_delete', pgrecs, logact) - return sret # # update one or multiple rows in wfile and/or wfile_dsid for multiple dsid # exclude dsid in condition # -def pgupdt_wfile_dsids(dsid, dsids, brec, bcnd, logact = PgLOG.LOGERR): - +def pgupdt_wfile_dsids(dsid, dsids, brec, bcnd, logact=None): + if logact is None: logact = PgLOG.LOGERR record = trim_wfile_fields(brec) if record: wret = PgDBI.pgupdt("wfile", record, bcnd, logact) @@ -241,15 +231,14 @@ def pgupdt_wfile_dsids(dsid, dsids, brec, bcnd, logact = PgLOG.LOGERR): if dsids: dids.extend(dsids.split(',')) for did in dids: wret += PgDBI.pgupdt("wfile_" + did, record, bcnd, logact|PgLOG.ADDTBL) - return wret # # get one record from wfile or wfile_dsid # exclude dsid in fields and condition # -def pgget_wfile(dsid, fields, condition, logact = PgLOG.LOGERR): - +def pgget_wfile(dsid, fields, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR tname = "wfile_" + dsid flds = fields.replace('wfile.', tname + '.') cnd = condition.replace('wfile.', tname + '.') @@ -261,8 +250,8 @@ def pgget_wfile(dsid, fields, condition, logact = PgLOG.LOGERR): # get one record from wfile or wfile_dsid joing other tables # exclude dsid in fields and condition # -def pgget_wfile_join(dsid, tjoin, fields, condition, logact = PgLOG.LOGERR): - +def pgget_wfile_join(dsid, tjoin, fields, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR tname = "wfile_" + dsid flds = fields.replace('wfile.', tname + '.') jname = tname + ' ' + tjoin.replace('wfile.', tname + '.') @@ -275,8 +264,8 @@ def pgget_wfile_join(dsid, tjoin, fields, condition, logact = PgLOG.LOGERR): # get multiple records from wfile or wfile_dsid # exclude dsid in fields and condition # -def pgmget_wfile(dsid, fields, condition, logact = PgLOG.LOGERR): - +def pgmget_wfile(dsid, fields, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR tname = "wfile_" + dsid flds = fields.replace('wfile.', tname + '.') cnd = condition.replace('wfile.', tname + '.') @@ -288,8 +277,8 @@ def pgmget_wfile(dsid, fields, condition, logact = PgLOG.LOGERR): # get multiple records from wfile or wfile_dsid joining other tables # exclude dsid in fields and condition # -def pgmget_wfile_join(dsid, tjoin, fields, condition, logact = PgLOG.LOGERR): - +def pgmget_wfile_join(dsid, tjoin, fields, condition, logact=None): + if logact is None: logact = PgLOG.LOGERR tname = "wfile_" + dsid flds = fields.replace('wfile.', tname + '.') jname = tname + ' ' + tjoin.replace('wfile.', tname + '.') diff --git a/src/rda_python_common/PgUtil.py b/src/rda_python_common/PgUtil.py index 38a1915..278b09a 100644 --- a/src/rda_python_common/PgUtil.py +++ b/src/rda_python_common/PgUtil.py @@ -18,6 +18,8 @@ import datetime import calendar import glob +import bisect +import functools from os import path as op from . import PgLOG @@ -61,25 +63,23 @@ def get_weekday(date = None): # Return: numeric Month if not fmt (default); three-charater or full month names for given fmt # def get_month(mn, fmt = None): - if not isinstance(mn, int): - if re.match(r'^\d+$', mn): + if mn.isdigit(): mn = int(mn) else: for m in range(12): if re.match(mn, MONTHS[m], re.I): mn = m + 1 break - - if fmt and mn > 0 and mn < 13: + if fmt and 0 < mn < 13: slen = len(fmt) if slen == 2: smn = "{:02}".format(mn) - elif re.match(r'^mon', fmt, re.I): + elif fmt[:3].lower() == 'mon': smn = MNS[mn-1] if slen == 3 else MONTHS[mn-1] - if re.match(r'^Mon', fmt): + if fmt.startswith('Mon'): smn = smn.capitalize() - elif re.match(r'^MON', fmt): + elif fmt.startswith('MON'): smn = smn.upper() else: smn = str(mn) @@ -92,29 +92,27 @@ def get_month(mn, fmt = None): # Return: numeric Weekday if !fmt (default); three-charater or full week name for given fmt # def get_wday(wday, fmt = None): - if not isinstance(wday, int): - if re.match(r'^\d+$', wday): + if wday.isdigit(): wday = int(wday) else: for w in range(7): if re.match(wday, WDAYS[w], re.I): wday = w break - - if fmt and wday >= 0 and wday <= 6: + if fmt and 0 <= wday <= 6: slen = len(fmt) if slen == 4: - swday = WDAYS[w] - if re.match(r'^We', fmt): + swday = WDAYS[wday] + if fmt.startswith('We'): swday = swday.capitalize() - elif re.match(r'^WE', fmt): + elif fmt.startswith('WE'): swday = swday.upper() elif slen == 3: swday = WDS[wday] - if re.match(r'^Ww', fmt): + if fmt.startswith('Ww'): swday = swday.capitalize() - elif re.match(r'^WW', fmt): + elif fmt.startswith('WW'): swday = swday.upper() else: swday = str(wday) @@ -127,19 +125,13 @@ def get_wday(wday, fmt = None): # Return: type if given file name is a valid online file; '' otherwise # def valid_online_file(file, type = None, exists = None): - if exists is None or exists: if not op.exists(file): return '' # file does not exist - bname = op.basename(file) - if re.match(r'^,.*', bname): return '' # hidden file - + if bname.startswith(','): return '' # hidden file if re.search(r'index\.(htm|html|shtml)$', bname, re.I): return '' # index file - if type and type != 'D': return type - if re.search(r'\.(doc|php|html|shtml)(\.|$)', bname, re.I): return '' # file with special extention - return 'D' # @@ -176,9 +168,8 @@ def curdatehour(fmt = None): # Return: current date and time strings # def get_date_time(tm = None): - act = ct = None - if tm == None: + if tm is None: ct = time.gmtime() if PgLOG.PGLOG['GMTZ'] else time.localtime() elif isinstance(tm, str): act = tm.split(' ') @@ -190,8 +181,7 @@ def get_date_time(tm = None): act = [str(tm), '00:00:00'] elif isinstance(tm, datetime.time): act = [None, str(tm)] - - if ct == None: + if ct is None: return act if act else None else: return [time.strftime("%Y-%m-%d", ct), time.strftime("%H:%M:%S", ct)] @@ -201,8 +191,7 @@ def get_date_time(tm = None): # Return: current datetime strings # def get_datetime(tm = None): - - if tm == None: + if tm is None: ct = time.gmtime() if PgLOG.PGLOG['GMTZ'] else time.localtime() return time.strftime("%Y-%m-%d %H:%M:%S", ct) elif isinstance(tm, str): @@ -214,7 +203,6 @@ def get_datetime(tm = None): return str(tm) elif isinstance(tm, datetime.date): return (str(tm) + ' 00:00:00') - return tm @@ -237,11 +225,9 @@ def timestamp(file = None): # check date/time and set to default one if empty date # def check_datetime(date, default): - if not date: return default if not isinstance(date, str): date = str(date) - if re.match(r'^0000', date): return default - + if date.startswith('0000'): return default return date # @@ -315,13 +301,8 @@ def format_datehour(date, hour, tofmt = None, fromfmt = None): # the sep value; str to int for digital values # def split_datetime(sdt, sep = r'\D'): - if not isinstance(sdt, str): sdt = str(sdt) - adt = re.split(sep, sdt) - acnt = len(adt) - for i in range(acnt): - if re.match(r'^\d+$', adt[i]): adt[i] = int(adt[i]) - return adt + return [int(x) if x.isdigit() else x for x in re.split(sep, sdt)] # # date: given date in format of fromfmt @@ -330,7 +311,6 @@ def split_datetime(sdt, sep = r'\D'): # Return: new formated date string according to tofmt # def format_date(cdate, tofmt = None, fromfmt = None): - if not cdate: return cdate if not isinstance(cdate, str): cdate = str(cdate) dates = [None, None, None] @@ -340,7 +320,6 @@ def format_date(cdate, tofmt = None, fromfmt = None): mkeys = ['D', 'M', 'Q', 'Y', 'C', 'H'] PATTERNS = [r'(\d\d\d\d)', r'(\d+)', r'(\d\d)', r'(\d\d\d)', '(' + mns + ')', '(' + months + ')'] - if not fromfmt: if not tofmt: if re.match(r'^\d\d\d\d-\d\d-\d\d$', cdate): return cdate # no need formatting @@ -349,7 +328,6 @@ def format_date(cdate, tofmt = None, fromfmt = None): fromfmt = "Y" + ms.group(1) + "M" + ms.group(2) + "D" else: PgLOG.pglog(cdate + ": Invalid date, should be in format YYYY-MM-DD", PgLOG.LGEREX) - pattern = fromfmt fmts = {} formats = {} @@ -358,7 +336,6 @@ def format_date(cdate, tofmt = None, fromfmt = None): if ms: fmts[mkey] = ms.group(1) pattern = re.sub(fmts[mkey], '', pattern) - cnt = 0 for mkey in fmts: fmt = fmts[mkey] @@ -371,8 +348,7 @@ def format_date(cdate, tofmt = None, fromfmt = None): if i == 4: i = 0 formats[fromfmt.find(fmt)] = fmt fromfmt = fromfmt.replace(fmt, PATTERNS[i]) - cnt += 1 - + cnt += 1 ms = re.findall(fromfmt, cdate) mcnt = len(ms[0]) if ms else 0 i = 0 @@ -380,29 +356,28 @@ def format_date(cdate, tofmt = None, fromfmt = None): if i >= mcnt: break fmt = formats[k] val = ms[0][i] - if re.match(r'^Y', fmt, re.I): + head = fmt[:1].upper() + if head == 'Y': dates[0] = int(val) if len(fmt) == 3: dates[0] *= 10 - elif re.match(r'^C', fmt, re.I): + elif head == 'C': dates[0] = 100 * int(val) # year at end of century - elif re.match(r'^M', fmt, re.I): - if re.match(r'^Mon', fmt, re.I): + elif head == 'M': + if fmt[:3].upper() == 'MON': dates[1] = get_month(val) else: dates[1] = int(val) - elif re.match(r'^Q', fmt, re.I): + elif head == 'Q': dates[1] = 3 * int(val) # month at end of quarter - elif re.match(r'^H', fmt, re.I): # hour + elif head == 'H': # hour dates.append(int(val)) else: # day dates[2] = int(val) - i += 1 - + i += 1 if len(dates) > 3: cdate = fmtdatehour(dates[0], dates[1], dates[2], dates[3], tofmt) else: cdate = fmtdate(dates[0], dates[1], dates[2], tofmt) - return cdate # @@ -416,27 +391,16 @@ def format_date(cdate, tofmt = None, fromfmt = None): # Return: new formated datehour string # def fmtdatetime(yr, mn, dy, hr = None, nn = None, ss = None, tofmt = None): - if not tofmt: tofmt = "YYYY-MM-DD HH:NN:SS" - tms = [ss, nn, hr, dy] fks = ['S', 'N', 'H'] ups = [60, 60, 24] - # adjust second/minute/hour values out of range for i in range(3): if tms[i] != None and tms[i+1] != None: - if tms[i] < 0: - while tms[i] < 0: - tms[i] += ups[i] - tms[i+1] -= 1 - elif tms[i] >= ups[i]: - while tms[i] >= ups[i]: - tms[i] -= ups[i] - tms[i+1] += 1 - + carry, tms[i] = divmod(tms[i], ups[i]) + tms[i+1] += carry sdt = fmtdate(yr, mn, dy, tofmt) - # format second/minute/hour values for i in range(3): if tms[i] != None: @@ -444,11 +408,10 @@ def fmtdatetime(yr, mn, dy, hr = None, nn = None, ss = None, tofmt = None): if ms: fmt = ms.group(1) if len(fmt) == 2: - str = "{:02}".format(tms[i]) + sval = "{:02}".format(tms[i]) else: - str = str(tms[i]) - sdt = re.sub(fmt, str, sdt, 1) - + sval = str(tms[i]) + sdt = re.sub(fmt, sval, sdt, 1) return sdt # @@ -460,21 +423,11 @@ def fmtdatetime(yr, mn, dy, hr = None, nn = None, ss = None, tofmt = None): # Return: new formated datehour string # def fmtdatehour(yr, mn, dy, hr, tofmt = None): - if not tofmt: tofmt = "YYYY-MM-DD:HH" - if hr != None and dy != None: # adjust hour value out of range - if hr < 0: - while hr < 0: - hr += 24 - dy -= 1 - elif hr > 23: - while hr > 23: - hr -= 24 - dy += 1 - + carry, hr = divmod(hr, 24) + dy += carry datehour = fmtdate(yr, mn, dy, tofmt) - if hr != None: ms = re.search(DATEFMTS['H'], datehour, re.I) if ms: @@ -484,7 +437,6 @@ def fmtdatehour(yr, mn, dy, hr, tofmt = None): else: shr = str(hr) datehour = re.sub(fmt, shr, datehour, 1) - return datehour # @@ -495,10 +447,8 @@ def fmtdatehour(yr, mn, dy, hr, tofmt = None): # Return: new formated date string # def fmtdate(yr, mn, dy, tofmt = None): - (y, m, d) = adjust_ymd(yr, mn, dy) if not tofmt or tofmt == 'YYYY-MM-DD': return "{}-{:02}-{:02}".format(y, m, d) - if dy != None: md = re.search(DATEFMTS['D'], tofmt, re.I) if md: @@ -512,7 +462,6 @@ def fmtdate(yr, mn, dy, tofmt = None): else: sdy = str(d) tofmt = re.sub(fmt, sdy, tofmt, 1) - if mn != None: md = re.search(DATEFMTS['M'], tofmt, re.I) if md: @@ -520,11 +469,11 @@ def fmtdate(yr, mn, dy, tofmt = None): slen = len(fmt) if slen == 2: smn = "{:02}".format(m) - elif re.match(r'^mon', fmt, re.I): + elif fmt[:3].lower() == 'mon': smn = MNS[m-1] if slen == 3 else MONTHS[m-1] - if re.match(r'^Mo', fmt): + if fmt.startswith('Mo'): smn = smn.capitalize() - elif re.match(r'^MO', fmt): + elif fmt.startswith('MO'): smn = smn.upper() else: smn = str(m) @@ -536,7 +485,6 @@ def fmtdate(yr, mn, dy, tofmt = None): m = int((m+2)/3) smn = "{:02}".format(m) if len(fmt) == 2 else str(m) tofmt = re.sub(fmt, smn, tofmt, 1) - if yr != None: md = re.search(DATEFMTS['Y'], tofmt, re.I) if md: @@ -562,7 +510,6 @@ def fmtdate(yr, mn, dy, tofmt = None): y = 1 + int(yr/10) syr = "{:02}".format(y) tofmt = re.sub(fmt, syr, tofmt, 1) - return tofmt # @@ -584,12 +531,10 @@ def join_datetime(sdate, stime): # split a date or datetime into an array of [date, time] # def date_and_time(sdt): - if not sdt: return [None, None] if not isinstance(sdt, str): sdt = str(sdt) - adt = re.split(' ', sdt) - acnt = len(adt) - if acnt == 1: adt.append('00:00:00') + adt = sdt.split(' ') + if len(adt) == 1: adt.append('00:00:00') return adt # @@ -707,8 +652,9 @@ def format_period(sdate, edate, fmt = None): # newid: True to format a new dsid; defaults to False for now # returns a new or old dsid according to the newid option # -def format_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGEREX): - +def format_dataset_id(dsid, newid = None, logact = None): + if newid is None: newid = PgLOG.PGLOG['NEWDSID'] + if logact is None: logact = PgLOG.LGEREX dsid = str(dsid) ms = re.match(r'^([a-z])(\d\d\d)(\d\d\d)$', dsid) if ms: @@ -721,7 +667,6 @@ def format_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGERE if logact: PgLOG.pglog(dsid + ": Cannot convert new dsid to old format", logact) return dsid return 'ds{}.{}'.format(ids[1], ids[2][2]) - ms = re.match(r'^ds(\d\d\d)(\.|)(\d)$', dsid, re.I) if not ms: ms = re.match(r'^(\d\d\d)(\.)(\d)$', dsid) if ms: @@ -729,7 +674,6 @@ def format_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGERE return "d{}00{}".format(ms.group(1), ms.group(3)) else: return 'ds{}.{}'.format(ms.group(1), ms.group(3)) - if logact: PgLOG.pglog(dsid + ": invalid dataset id", logact) return dsid @@ -738,8 +682,9 @@ def format_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGERE # newid: True to format a new dsid; defaults to False for now # returns a new or old metadata dsid according to the newid option # -def metadata_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGEREX): - +def metadata_dataset_id(dsid, newid = None, logact = None): + if newid is None: newid = PgLOG.PGLOG['NEWDSID'] + if logact is None: logact = PgLOG.LGEREX ms = re.match(r'^([a-z])(\d\d\d)(\d\d\d)$', dsid) if ms: ids = list(ms.groups()) @@ -751,7 +696,6 @@ def metadata_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGE if logact: PgLOG.pglog(dsid + ": Cannot convert new dsid to old format", logact) return dsid return '{}.{}'.format(ids[1], ids[2][2]) - ms = re.match(r'^ds(\d\d\d)(\.|)(\d)$', dsid) if not ms: ms = re.match(r'^(\d\d\d)(\.)(\d)$', dsid) if ms: @@ -759,7 +703,6 @@ def metadata_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGE return "d{}00{}".format(ms.group(1), ms.group(3)) else: return '{}.{}'.format(ms.group(1), ms.group(3)) - if logact: PgLOG.pglog(dsid + ": invalid dataset id", logact) return dsid @@ -770,7 +713,6 @@ def metadata_dataset_id(dsid, newid = PgLOG.PGLOG['NEWDSID'], logact = PgLOG.LGE # returns dsid if found in given id string; None otherwise # def find_dataset_id(idstr, flag = 'B', logact = 0): - if flag in 'NB': ms = re.search(r'(^|\W)(([a-z])\d{6})($|\D)', idstr) if ms and ms.group(3) in PgLOG.PGLOG['DSIDCHRS']: return ms.group(2) @@ -778,16 +720,15 @@ def find_dataset_id(idstr, flag = 'B', logact = 0): ms = re.search(r'(^|\W)(ds\d\d\d(\.|)\d)($|\D)', idstr) if not ms: ms = re.search(r'(^|\W)(\d\d\d\.\d)($|\D)', idstr) if ms: return ms.group(2) - - if logact: PgLOG.pglog("{} : No valid dsid found for flag {}".format(idstr, flag), logact) + if logact: PgLOG.pglog("{}: No valid dsid found for flag {}".format(idstr, flag), logact) return None # # find and convert all found dsids according to old/new dsids # for newid = False/True # -def convert_dataset_ids(idstr, newid = PgLOG.PGLOG['NEWDSID'], logact = 0): - +def convert_dataset_ids(idstr, newid = None, logact = 0): + if newid is None: newid = PgLOG.PGLOG['NEWDSID'] flag = 'O' if newid else 'N' cnt = 0 if idstr: @@ -797,7 +738,6 @@ def convert_dataset_ids(idstr, newid = PgLOG.PGLOG['NEWDSID'], logact = 0): ndsid = format_dataset_id(dsid, newid = newid, logact = logact) if ndsid != dsid: idstr = idstr.replace(dsid, ndsid) cnt += 1 - return (idstr, cnt) # @@ -950,22 +890,16 @@ def joinhash(adict, bdict, default = None, unique = None): # Return: the joined list # def joinarray(lst1, lst2, unique = None): - if not lst2: return lst1 if not lst1: return lst2 - cnt1 = len(lst1) cnt2 = len(lst2) - if unique: - for i in (cnt2): - for j in (cnt1): - if pgcmp(lst1[j], lst2[i]) != 0: break - if j >= cnt1: - lst1.append(lst2[i]) + for i in range(cnt2): + if lst2[i] not in lst1: + lst1.append(lst2[i]) else: lst1.extend(lst2) - return lst1 # @@ -1008,9 +942,7 @@ def strip_field(field): # Return: a sorted dict list # def sorthash(pgrecs, flds, hash, patterns = None): - fcnt = len(flds) # count of fields to be sorted on - # set sorting order, descenting (-1) or ascenting (1) # get the full field names to be sorted on desc = [1]*fcnt @@ -1020,12 +952,9 @@ def sorthash(pgrecs, flds, hash, patterns = None): if flds[i].islower(): desc[i] = -1 fld = strip_field(hash[flds[i].upper()][1]) fields.append(fld) - count = len(pgrecs[fields[0]]) # row count of pgrecs - if count < 2: return pgrecs # no need of sording pcnt = len(patterns) if patterns else 0 - # prepare the dict list for sortting srecs = [] for i in range(count): @@ -1038,35 +967,33 @@ def sorthash(pgrecs, flds, hash, patterns = None): else: # sort on the whole value if no pattern given val = pgrec[fields[j]] - if nums[j]: nums[j] = pgnum(val) rec.append(val) rec.append(i) # add column to cache the row index srecs.append(rec) - - srecs = quicksort(srecs, 0, count-1, desc, fcnt, nums) - + srecs.sort(key=functools.cmp_to_key( + lambda a, b: cmp_records(a, b, desc, fcnt, nums))) # sort pgrecs according the cached row index column in ordered srecs rets = {} for fld in pgrecs: rets[fld] = [] - for i in range(count): pgrec = onerecord(pgrecs, srecs[i][fcnt]) for fld in pgrecs: rets[fld].append(pgrec[fld]) - return rets # # Return: the number of days bewteen date1 and date2 # def diffdate(date1, date2): - - ut1 = ut2 = 0 - if date1: ut1 = unixtime(date1) - if date2: ut2 = unixtime(date2) - return round((ut1 - ut2)/86400) # 24*60*60 + epoch = datetime.date(1970, 1, 1) + def _to_date(d): + if not d: return epoch + ms = re.match(r'^(\d+)-(\d+)-(\d+)', str(d)) + if not ms: return epoch + return datetime.date(int(ms.group(1)), int(ms.group(2)), int(ms.group(3))) + return (_to_date(date1) - _to_date(date2)).days # # Return: the number of seconds bewteen time1 and time2 @@ -1212,26 +1139,17 @@ def addmonth(cdate, mf, nf = 1): # add yr years & mn months to yearmonth ym in format YYYYMM def addyearmonth(ym, yr, mn): - - if yr == None: yr = 0 - if mn == None: mn = 0 - + if yr is None: yr = 0 + if mn is None: mn = 0 ms =re.match(r'^(\d\d\d\d)(\d\d)$', ym) if ms: (syr, smn) = ms.groups() - yr = int(syr) - mn = int(smn) - if mn < 0: - while mn < 0: - yr -= 1 - mn += 12 - else: - while mn > 12: - yr += 1 - mn -= 12 - - ym = "{:04}{:02}".format(yr, mn) - + nyr = int(syr) + yr + nmn = int(smn) + mn - 1 # shift to 0-indexed for divmod + extra, nmn = divmod(nmn, 12) + nyr += extra + nmn += 1 # back to 1-indexed + ym = "{:04}{:02}".format(nyr, nmn) return ym # @@ -1354,54 +1272,34 @@ def adddate(cdate, yr, mn = 0, dy = 0, tofmt = None): # add given hours to the initial date and time # def addhour(sdate, stime, nhour): - if nhour and isinstance(nhour, str): nhour = int(nhour) if sdate and not isinstance(sdate, str): sdate = str(sdate) if stime and not isinstance(stime, str): stime = str(stime) if not nhour: return [sdate, stime] - hr = dy = 0 ms = re.match(r'^(\d+)', stime) if ms: shr = ms.group(1) hr = int(shr) + nhour - if hr < 0: - while hr < 0: - dy -= 1 - hr += 24 - else: - while hr > 23: - dy += 1 - hr -= 24 - + dy, hr = divmod(hr, 24) shour = "{:02}".format(hr) if shr != shour: stime = re.sub(shr, shour, stime, 1) if dy: sdate = adddate(sdate, 0, 0, dy) - return [sdate, stime] # # add given years, months, days and hours to the initial date and hour # def adddatehour(sdate, nhour, yr, mn, dy, hr = 0): - if sdate and not isinstance(sdate, str): sdate = str(sdate) if hr: if nhour != None: if isinstance(nhour, str): nhour = int(nhour) hr += nhour - if hr < 0: - while hr < 0: - dy -= 1 - hr += 24 - else: - while hr > 23: - dy += 1 - hr -= 24 + carry, hr = divmod(hr, 24) + dy += carry if nhour != None: nhour = hr - if yr or mn or dy: sdate = adddate(sdate, yr, mn, dy) - return [sdate, nhour] # @@ -1426,28 +1324,23 @@ def adddatetime(sdatetime, yy, mm, dd, hh, nn, ss, nf = 0): # if nf, add fraction of month only # def adddatetime(sdatetime, yy, mm, dd, hh, nn, ss, nf = 0): - if sdatetime and not isinstance(sdatetime, str): sdatetime = str(sdatetime) - (sdate, stime) = re.split(' ', sdatetime) - + (sdate, stime) = sdatetime.split(' ', 1) if hh or nn or ss: (sdate, stime) = addtime(sdate, stime, hh, nn, ss) if nf: sdate = addmonth(sdate, mm, nf) mm = 0 if yy or mm or dd: sdate = adddate(sdate, yy, mm, dd) - return "{} {}".format(sdate, stime) # # add given hours, minutes and seconds to the initial date and time # def addtime(sdate, stime, h, m, s): - if sdate and not isinstance(sdate, str): sdate = str(sdate) - if stime and not isinstance(stime, str): sdate = str(stime) + if stime and not isinstance(stime, str): stime = str(stime) ups = (60, 60, 24) tms = [0, 0, 0, 0] # (sec, min, hour, day) - if s: tms[0] += s if m: tms[1] += m if h: tms[2] += h @@ -1457,20 +1350,11 @@ def addtime(sdate, stime, h, m, s): tms[2] += int(ms.group(1)) tms[1] += int(ms.group(2)) tms[0] += int(ms.group(3)) - for i in range(3): - if tms[i] < 0: - while tms[i] < 0: - tms[i] += ups[i] - tms[i+1] -= 1 - elif tms[i] >= ups[i]: - while tms[i] >= ups[i]: - tms[i] -= ups[i] - tms[i+1] += 1 - + carry, tms[i] = divmod(tms[i], ups[i]) + tms[i+1] += carry stime = "{:02}:{:02}:{:02}".format(tms[2], tms[1], tms[0]) if tms[3]: sdate = adddate(sdate, 0, 0, tms[3]) - return [sdate, stime] # @@ -1559,15 +1443,12 @@ def enddate(sdate, days, unit, nf = 0): # adjust end time to the specified h/n/s for frequency of hour/mimute/second # def endtime(stime, unit): - if stime and not isinstance(stime, str): stime = str(stime) - if not (unit and unit in 'HNS'): return stime - + if not (unit and unit in 'HNS'): return stime if stime: - tm = split_datetime(stime, 'T') + tm = split_datetime(stime) # split on non-digits to get [HH, MM, SS] else: tm = [0, 0, 0] - if unit == 'H': tm[1] = tm[2] = 59 elif unit == 'N': @@ -1575,18 +1456,15 @@ def endtime(stime, unit): elif unit != 'S': tm[0] = 23 tm[1] = tm[2] = 59 - - return "{:02}:{:02}:{:02}".format(tm[0], tm[1]. tm[2]) + return "{:02}:{:02}:{:02}".format(tm[0], tm[1], tm[2]) # # adjust end time to the specified h/n/s for frequency of year/month/week/day/hour/mimute/second # def enddatetime(sdatetime, unit, days = 0, nf = 0): - if sdatetime and not isinstance(sdatetime, str): sdatetime = str(sdatetime) if not (unit and unit in 'YMWDHNS'): return sdatetime - (sdate, stime) = re.split(' ', sdatetime) - + (sdate, stime) = sdatetime.split(' ', 1) if unit in 'HNS': stime = endtime(stime, unit) else: @@ -1597,16 +1475,13 @@ def enddatetime(sdatetime, unit, days = 0, nf = 0): # get the string length dynamically # def get_column_length(colname, values): - clen = len(colname) if colname else 2 # initial column length as the length of column title - for val in values: if val is None: continue sval = str(val) - if sval and not re.search(r'\n', sval): + if sval and '\n' not in sval: slen = len(sval) if slen > clen: clen = slen - return clen # @@ -1740,23 +1615,8 @@ def recursive_files(infiles): # Return: index if found; -1 otherwise # def asearch(lidx, hidx, key, list): - - ret = -1 - if (hidx - lidx) < 11: # use linear search for less than 11 items - for midx in range(lidx, hidx): - if key == list[midx]: - ret = midx - break - else: - midx = (lidx + hidx)/2 - if key == list[midx]: - ret = midx - elif key < list[midx]: - ret = asearch(lidx, midx, key, list) - else: - ret = asearch(midx + 1, hidx, key, list) - - return ret + idx = bisect.bisect_left(list, key, lidx, hidx) + return idx if idx < hidx and list[idx] == key else -1 # # lidx: lower index limit (including) @@ -1844,11 +1704,23 @@ def format_float_value(val, precision = 2): # check a file is a ASCII text one # return 1 if yes, 0 if not; or -1 if file not exists # -def is_text_file(fname): - - ret = -1 - if op.isfile(fname): - buf = PgLOG.pgsystem("file -b " + fname, PgLOG.LOGWRN, 20) - ret = 1 if buf and re.search(r'(^|\s)(text|script|data)', buf) else 0 - - return ret +def is_text_file(fname, blocksize = 256, threshhold = 0.1): + # File doesn't exist or is not a regular file + if not op.exists(fname) or not op.isfile(fname): return -1 + if op.getsize(fname) == 0: return 1 # Empty files are considered text + try: + buffer = None + with open(fname, 'rb') as f: + buffer = f.read(blocksize) + # Check for null bytes (a strong indicator of a binary file) + if not buffer or b'\0' in buffer: return 0 + text_set = frozenset( + b'\t\n\r\f\v' + # Whitespace characters + bytes(range(32, 127)) # Printable ASCII characters + ) + non_text_count = sum(b not in text_set for b in buffer) + # If a significant portion of the buffer consists of non-text characters, + # it's likely a binary file. + return 1 if((non_text_count/len(buffer)) < threshhold) else 0 + except IOError: + return -1 # Handle cases where the file cannot be opened or read diff --git a/src/rda_python_common/__init__.py b/src/rda_python_common/__init__.py index a2a75e2..3877cdd 100644 --- a/src/rda_python_common/__init__.py +++ b/src/rda_python_common/__init__.py @@ -22,7 +22,7 @@ from . import PgLOG, PgUtil, PgDBI, PgFile, PgLock, PgCMD, PgSIG, PgOPT, PgSplit -__version__ = "3.0.13" +__version__ = "3.0.14" __all__ = [ "PgLOG", diff --git a/src/rda_python_common/pg_log.py b/src/rda_python_common/pg_log.py index c685783..e6e878b 100644 --- a/src/rda_python_common/pg_log.py +++ b/src/rda_python_common/pg_log.py @@ -1001,14 +1001,14 @@ def seconds_to_string_time(seconds, showzero=0): msg = "0S" return msg - def tosystem(self, cmd, timeout=0, logact=0, cmdopt=5, instr=None): + def tosystem(self, cmd, timeout=0, logact=None, cmdopt=5, instr=None): """Run a system command with a timeout via :meth:`pgsystem`. Args: cmd: Command string or list (passed to :meth:`pgsystem`). timeout: Seconds before the command is killed. Uses ``PGLOG['TIMEOUT']`` when 0. - logact: Logging action flags. + logact: Logging action flags; defaults to ``LOGWRN``. cmdopt: Command option bitfield (see :meth:`pgsystem`). instr: String passed to the command via stdin. diff --git a/src/rda_python_common/pg_util.py b/src/rda_python_common/pg_util.py index a449664..419f045 100644 --- a/src/rda_python_common/pg_util.py +++ b/src/rda_python_common/pg_util.py @@ -1919,7 +1919,7 @@ def endtime(self, stime, unit): if stime and not isinstance(stime, str): stime = str(stime) if not (unit and unit in 'HNS'): return stime if stime: - tm = self.split_datetime(stime, 'T') + tm = self.split_datetime(stime) # split on non-digits to get [HH, MM, SS] else: tm = [0, 0, 0] if unit == 'H':