extract_oracle_models.py script, for legacy Oracle 11g databases

This commit is contained in:
Oscar Fonts
2013-11-19 19:27:34 +01:00
parent b05daddc5c
commit 99a6dbeac2
+144 -118
View File
@@ -1,42 +1,41 @@
#!/usr/bin/env python #!/usr/bin/env python
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
"""Create web2py model (python code) to represent PostgreSQL tables. """Create web2py model (python code) to represent Oracle 11g tables.
Features: Features:
* Uses ANSI Standard INFORMATION_SCHEMA (might work with other RDBMS) * Uses Oracle's metadata tables
* Detects legacy "keyed" tables (not having an "id" PK) * Detects legacy "keyed" tables (not having an "id" PK)
* Connects directly to running databases, no need to do a SQL dump * Connects directly to running databases, no need to do a SQL dump
* Handles notnull, unique and referential constraints * Handles notnull, unique and referential constraints
* Detects most common datatypes and default values * Detects most common datatypes and default values
* Support PostgreSQL columns comments (ie. for documentation) * Documents alternative datatypes as comments
Requeriments: Requeriments:
* Needs PostgreSQL pyscopg2 python connector (same as web2py) * Needs Oracle cx_Oracle python connector (same as web2py)
* If used against other RDBMS, import and use proper connector (remove pg_ code)
Created by Mariano Reingart, based on a script to "generate schemas from dbs" Created by Oscar Fonts, based on extract_pgsql_models by Mariano Reingart,
(mysql) by Alexandre Andrade based in turn on a script to "generate schemas from dbs" (mysql)
by Alexandre Andrade
""" """
_author__ = "Mariano Reingart <reingart@gmail.com>" _author__ = "Oscar Fonts <oscar.fonts@geomati.co>"
HELP = """ HELP = """
USAGE: extract_pgsql_models db host port user passwd USAGE: extract_oracle_models db host port user passwd
Call with PostgreSQL database connection parameters, Call with Oracle database connection parameters,
web2py model will be printed on standard output. web2py model will be printed on standard output.
EXAMPLE: python extract_pgsql_models.py mydb localhost 5432 reingart saraza EXAMPLE: python extract_oracle_models.py ORCL localhost 1521 user password
""" """
# Config options # Config options
DEBUG = False # print debug messages to STDERR DEBUG = False # print debug messages to STDERR
SCHEMA = 'public' # change if not using default PostgreSQL schema
# Constant for Field keyword parameter order (and filter): # Constant for Field keyword parameter order (and filter):
KWARGS = ('type', 'length', 'default', 'required', 'ondelete', KWARGS = ('type', 'length', 'default', 'required', 'ondelete',
@@ -52,7 +51,7 @@ def query(conn, sql, *args):
ret = [] ret = []
try: try:
if DEBUG: if DEBUG:
print >> sys.stderr, "QUERY: ", sql % args print >> sys.stderr, "QUERY: ", sql , args
cur.execute(sql, args) cur.execute(sql, args)
for row in cur: for row in cur:
dic = {} dic = {}
@@ -63,16 +62,18 @@ def query(conn, sql, *args):
print >> sys.stderr, "RET: ", dic print >> sys.stderr, "RET: ", dic
ret.append(dic) ret.append(dic)
return ret return ret
except cx_Oracle.DatabaseError, exc:
error, = exc.args
print >> sys.stderr, "Oracle-Error-Message:", error.message
finally: finally:
cur.close() cur.close()
def get_tables(conn, schema=SCHEMA): def get_tables(conn):
"List table names in a given schema" "List table names in a given schema"
rows = query(conn, """SELECT table_name FROM information_schema.tables rows = query(conn, """SELECT TABLE_NAME FROM USER_TABLES
WHERE table_schema = %s ORDER BY TABLE_NAME""")
ORDER BY table_name""", schema) return [row['TABLE_NAME'] for row in rows]
return [row['table_name'] for row in rows]
def get_fields(conn, table): def get_fields(conn, table):
@@ -80,154 +81,180 @@ def get_fields(conn, table):
if DEBUG: if DEBUG:
print >> sys.stderr, "Processing TABLE", table print >> sys.stderr, "Processing TABLE", table
rows = query(conn, """ rows = query(conn, """
SELECT column_name, data_type, SELECT COLUMN_NAME, DATA_TYPE,
is_nullable, NULLABLE AS IS_NULLABLE,
character_maximum_length, CHAR_LENGTH AS CHARACTER_MAXIMUM_LENGTH,
numeric_precision, numeric_precision_radix, numeric_scale, DATA_PRECISION AS NUMERIC_PRECISION,
column_default DATA_SCALE AS NUMERIC_SCALE,
FROM information_schema.columns DATA_DEFAULT AS COLUMN_DEFAULT
WHERE table_name=%s FROM USER_TAB_COLUMNS
ORDER BY ordinal_position""", table) WHERE TABLE_NAME=:t
""", table)
return rows return rows
def define_field(conn, table, field, pks): def define_field(conn, table, field, pks):
"Determine field type, default value, references, etc." "Determine field type, default value, references, etc."
f = {} f = {}
ref = references(conn, table, field['column_name']) ref = references(conn, table, field['COLUMN_NAME'])
# Foreign Keys
if ref: if ref:
f.update(ref) f.update(ref)
elif field['column_default'] and \ # PK & Numeric & autoincrement => id
field['column_default'].startswith("nextval") and \ elif field['COLUMN_NAME'] in pks and \
field['column_name'] in pks: field['DATA_TYPE'] in ('INT', 'NUMBER') and \
# postgresql sequence (SERIAL) and primary key! is_autoincrement(conn, table, field):
f['type'] = "'id'" f['type'] = "'id'"
elif field['data_type'].startswith('character'): # Other data types
f['type'] = "'string'" elif field['DATA_TYPE'] in ('BINARY_DOUBLE'):
if field['character_maximum_length']:
f['length'] = field['character_maximum_length']
elif field['data_type'] in ('text', ):
f['type'] = "'text'"
elif field['data_type'] in ('boolean', 'bit'):
f['type'] = "'boolean'"
elif field['data_type'] in ('integer', 'smallint', 'bigint'):
f['type'] = "'integer'"
elif field['data_type'] in ('double precision', 'real'):
f['type'] = "'double'" f['type'] = "'double'"
elif field['data_type'] in ('timestamp', 'timestamp without time zone'): elif field['DATA_TYPE'] in ('CHAR','NCHAR'):
f['type'] = "'datetime'" f['type'] = "'string'"
elif field['data_type'] in ('date', ): f['comment'] = "'Alternative types: boolean, time'"
f['type'] = "'date'" elif field['DATA_TYPE'] in ('BLOB', 'CLOB'):
elif field['data_type'] in ('time', 'time without time zone'):
f['type'] = "'time'"
elif field['data_type'] in ('numeric', 'currency'):
f['type'] = "'decimal'"
f['precision'] = field['numeric_precision']
f['scale'] = field['numeric_scale'] or 0
elif field['data_type'] in ('bytea', ):
f['type'] = "'blob'" f['type'] = "'blob'"
elif field['data_type'] in ('point', 'lseg', 'polygon', 'unknown', 'USER-DEFINED'): f['comment'] = "'Alternative types: text, json, list:*'"
f['type'] = "" # unsupported? elif field['DATA_TYPE'] in ('DATE'):
f['type'] = "'datetime'"
f['comment'] = "'Alternative types: date'"
elif field['DATA_TYPE'] in ('FLOAT'):
f['type'] = "'float'"
elif field['DATA_TYPE'] in ('INT'):
f['type'] = "'integer'"
elif field['DATA_TYPE'] in ('NUMBER'):
f['type'] = "'bigint'"
elif field['DATA_TYPE'] in ('NUMERIC'):
f['type'] = "'decimal'"
f['precision'] = field['NUMERIC_PRECISION']
f['scale'] = field['NUMERIC_SCALE'] or 0
elif field['DATA_TYPE'] in ('VARCHAR2','NVARCHAR2'):
f['type'] = "'string'"
if field['CHARACTER_MAXIMUM_LENGTH']:
f['length'] = field['CHARACTER_MAXIMUM_LENGTH']
f['comment'] = "'Other possible types: password, upload'"
else: else:
raise RuntimeError("Data Type not supported: %s " % str(field)) f['type'] = "'blob'"
f['comment'] = "'WARNING: Oracle Data Type %s was not mapped." % \
str(field['DATA_TYPE']) + " Using 'blob' as fallback.'"
try: try:
if field['column_default']: if field['COLUMN_DEFAULT']:
if field['column_default'] == "now()": if field['COLUMN_DEFAULT'] == "sysdate":
d = "request.now" d = "request.now"
elif field['column_default'] == "true": elif field['COLUMN_DEFAULT'].upper() == "T":
d = "True" d = "True"
elif field['column_default'] == "false": elif field['COLUMN_DEFAULT'].upper() == "F":
d = "False" d = "False"
else: else:
d = repr(eval(field['column_default'])) d = repr(eval(field['COLUMN_DEFAULT']))
f['default'] = str(d) f['default'] = str(d)
except (ValueError, SyntaxError): except (ValueError, SyntaxError):
pass pass
except Exception, e: except Exception, e:
raise RuntimeError( raise RuntimeError(
"Default unsupported '%s'" % field['column_default']) "Default unsupported '%s'" % field['COLUMN_DEFAULT'])
if not field['is_nullable']: if not field['IS_NULLABLE']:
f['notnull'] = "True" f['notnull'] = "True"
comment = get_comment(conn, table, field)
if comment is not None:
f['comment'] = repr(comment)
return f return f
def is_unique(conn, table, field): def is_unique(conn, table, field):
"Find unique columns (incomplete support)" "Find unique columns"
rows = query(conn, """ rows = query(conn, """
SELECT information_schema.constraint_column_usage.column_name SELECT COLS.COLUMN_NAME
FROM information_schema.table_constraints FROM USER_CONSTRAINTS CONS, ALL_CONS_COLUMNS COLS
NATURAL JOIN information_schema.constraint_column_usage WHERE CONS.OWNER = COLS.OWNER
WHERE information_schema.table_constraints.table_name=%s AND CONS.CONSTRAINT_NAME = COLS.CONSTRAINT_NAME
AND information_schema.constraint_column_usage.column_name=%s AND CONS.CONSTRAINT_TYPE = 'U'
AND information_schema.table_constraints.constraint_type='UNIQUE' AND COLS.TABLE_NAME = :t
;""", table, field['column_name']) AND COLS.COLUMN_NAME = :c
""", table, field['COLUMN_NAME'])
return rows and True or False return rows and True or False
def get_comment(conn, table, field): # Returns True when a "BEFORE EACH ROW INSERT" trigger is found and:
"Find the column comment (postgres specific)" # a) it mentions the "NEXTVAL" keyword (used by sequences)
# b) it operates on the given table and column
#
# On some (inelegant) database designs, SEQUENCE.NEXTVAL is called directly
# from each "insert" statement, instead of using triggers. Such cases cannot
# be detected by inspecting Oracle's metadata tables, as sequences are not
# logically bound to any specific table or field.
def is_autoincrement(conn, table, field):
"Find auto increment fields (best effort)"
rows = query(conn, """ rows = query(conn, """
SELECT d.description AS comment SELECT TRIGGER_NAME
FROM pg_class c FROM USER_TRIGGERS,
JOIN pg_description d ON c.oid=d.objoid (SELECT NAME, LISTAGG(TEXT, ' ') WITHIN GROUP (ORDER BY LINE) TEXT
JOIN pg_attribute a ON c.oid = a.attrelid FROM USER_SOURCE
WHERE c.relname=%s AND a.attname=%s WHERE TYPE = 'TRIGGER'
AND a.attnum = d.objsubid GROUP BY NAME
;""", table, field['column_name']) ) TRIGGER_DEFINITION
return rows and rows[0]['comment'] or None WHERE TRIGGER_NAME = NAME
AND TRIGGERING_EVENT = 'INSERT'
AND TRIGGER_TYPE = 'BEFORE EACH ROW'
AND TABLE_NAME = :t
AND UPPER(TEXT) LIKE UPPER('%.NEXTVAL%')
AND UPPER(TEXT) LIKE UPPER('%:NEW.' || :c || '%')
""", table, field['COLUMN_NAME'])
return rows and True or False
def primarykeys(conn, table): def primarykeys(conn, table):
"Find primary keys" "Find primary keys"
rows = query(conn, """ rows = query(conn, """
SELECT information_schema.constraint_column_usage.column_name SELECT COLS.COLUMN_NAME
FROM information_schema.table_constraints FROM USER_CONSTRAINTS CONS, ALL_CONS_COLUMNS COLS
NATURAL JOIN information_schema.constraint_column_usage WHERE COLS.TABLE_NAME = :t
WHERE information_schema.table_constraints.table_name=%s AND CONS.CONSTRAINT_TYPE = 'P'
AND information_schema.table_constraints.constraint_type='PRIMARY KEY' AND CONS.OWNER = COLS.OWNER
;""", table) AND CONS.CONSTRAINT_NAME = COLS.CONSTRAINT_NAME
return [row['column_name'] for row in rows] """, table)
return [row['COLUMN_NAME'] for row in rows]
def references(conn, table, field): def references(conn, table, field):
"Find a FK (fails if multiple)" "Find a FK (fails if multiple)"
rows1 = query(conn, """ rows1 = query(conn, """
SELECT table_name, column_name, constraint_name, SELECT COLS.CONSTRAINT_NAME,
update_rule, delete_rule, ordinal_position CONS.DELETE_RULE,
FROM information_schema.key_column_usage COLS.POSITION AS ORDINAL_POSITION
NATURAL JOIN information_schema.referential_constraints FROM USER_CONSTRAINTS CONS, ALL_CONS_COLUMNS COLS
NATURAL JOIN information_schema.table_constraints WHERE COLS.TABLE_NAME = :t
WHERE information_schema.key_column_usage.table_name=%s AND COLS.COLUMN_NAME = :c
AND information_schema.key_column_usage.column_name=%s AND CONS.CONSTRAINT_TYPE = 'R'
AND information_schema.table_constraints.constraint_type='FOREIGN KEY' AND CONS.OWNER = COLS.OWNER
;""", table, field) AND CONS.CONSTRAINT_NAME = COLS.CONSTRAINT_NAME
""", table, field)
if len(rows1) == 1: if len(rows1) == 1:
rows2 = query(conn, """ rows2 = query(conn, """
SELECT table_name, column_name, * SELECT COLS.TABLE_NAME, COLS.COLUMN_NAME
FROM information_schema.constraint_column_usage FROM USER_CONSTRAINTS CONS, ALL_CONS_COLUMNS COLS
WHERE constraint_name=%s WHERE CONS.CONSTRAINT_NAME = :k
""", rows1[0]['constraint_name']) AND CONS.R_CONSTRAINT_NAME = COLS.CONSTRAINT_NAME
ORDER BY COLS.POSITION ASC
""", rows1[0]['CONSTRAINT_NAME'])
row = None row = None
if len(rows2) > 1: if len(rows2) > 1:
row = rows2[int(rows1[0]['ordinal_position']) - 1] row = rows2[int(rows1[0]['ORDINAL_POSITION']) - 1]
keyed = True keyed = True
if len(rows2) == 1: if len(rows2) == 1:
row = rows2[0] row = rows2[0]
keyed = False keyed = False
if row: if row:
if keyed: # THIS IS BAD, DON'T MIX "id" and primarykey!!! if keyed: # THIS IS BAD, DON'T MIX "id" and primarykey!!!
ref = {'type': "'reference %s.%s'" % (row['table_name'], ref = {'type': "'reference %s.%s'" % (row['TABLE_NAME'],
row['column_name'])} row['COLUMN_NAME'])}
else: else:
ref = {'type': "'reference %s'" % (row['table_name'],)} ref = {'type': "'reference %s'" % (row['TABLE_NAME'],)}
if rows1[0]['delete_rule'] != "NO ACTION": if rows1[0]['DELETE_RULE'] != "NO ACTION":
ref['ondelete'] = repr(rows1[0]['delete_rule']) ref['ondelete'] = repr(rows1[0]['DELETE_RULE'])
return ref return ref
elif rows2: elif rows2:
raise RuntimeError("Unsupported foreign key reference: %s" % raise RuntimeError("Unsupported foreign key reference: %s" %
@@ -244,15 +271,15 @@ def define_table(conn, table):
pks = primarykeys(conn, table) pks = primarykeys(conn, table)
print "db.define_table('%s'," % (table, ) print "db.define_table('%s'," % (table, )
for field in fields: for field in fields:
fname = field['column_name'] fname = field['COLUMN_NAME']
fdef = define_field(conn, table, field, pks) fdef = define_field(conn, table, field, pks)
if fname not in pks and is_unique(conn, table, field): if fname not in pks and is_unique(conn, table, field):
fdef['unique'] = "True" fdef['unique'] = "True"
if fdef['type'] == "'id'" and fname in pks: if fdef['type'] == "'id'" and fname in pks:
pks.pop(pks.index(fname)) pks.pop(pks.index(fname))
print " Field('%s', %s)," % (fname, print " Field('%s', %s)," % (fname,
', '.join(["%s=%s" % (k, fdef[k]) for k in KWARGS ', '.join(["%s=%s" % (k, fdef[k]) for k in KWARGS
if k in fdef and fdef[k]])) if k in fdef and fdef[k]]))
if pks: if pks:
print " primarykey=[%s]," % ", ".join(["'%s'" % pk for pk in pks]) print " primarykey=[%s]," % ", ".join(["'%s'" % pk for pk in pks])
print " migrate=migrate)" print " migrate=migrate)"
@@ -261,7 +288,7 @@ def define_table(conn, table):
def define_db(conn, db, host, port, user, passwd): def define_db(conn, db, host, port, user, passwd):
"Output database definition (model)" "Output database definition (model)"
dal = 'db = DAL("postgres://%s:%s@%s:%s/%s", pool_size=10)' dal = 'db = DAL("oracle://%s/%s@%s:%s/%s", pool_size=10)'
print dal % (user, passwd, host, port, db) print dal % (user, passwd, host, port, db)
print print
print "migrate = False" print "migrate = False"
@@ -278,9 +305,8 @@ if __name__ == "__main__":
db, host, port, user, passwd = sys.argv[1:6] db, host, port, user, passwd = sys.argv[1:6]
# Make the database connection (change driver if required) # Make the database connection (change driver if required)
import psycopg2 import cx_Oracle
cnn = psycopg2.connect(database=db, host=host, port=port, dsn = cx_Oracle.makedsn(host, port, db)
user=user, password=passwd, cnn = cx_Oracle.connect(user, passwd, dsn)
)
# Start model code generation: # Start model code generation:
define_db(cnn, db, host, port, user, passwd) define_db(cnn, db, host, port, user, passwd)