#!/usr/bin/env python

"""
Copyright (c) 2006-2026 sqlmap developers (https://sqlmap.org)
See the file 'LICENSE' for copying permission
"""

from .engine import Esperanto
from .records import OracleUndecided


def _sanitize(s):
    """Escape C0/DEL/C1 control bytes in recovered DB content before it is logged - a stored
    value can carry ANSI/OSC sequences that would otherwise drive or forge the sqlmap console."""
    if not isinstance(s, type(u"")):
        return s
    return "".join(c if (u" " <= c < u"\x7f") or c > u"\x9f" else "\\x%02x" % ord(c) for c in s)


def buildHandler():
    """Build the sqlmap dbmsHandler that drives enumeration through this engine when
    the back-end cannot be (or should not be) fingerprinted. sqlmap-core imports are
    deferred here so the engine above stays dependency-free for standalone use.

    The user still commands *what* to retrieve (--banner / --tables / --dump / ...);
    esperanto only works out *how* on a dialect it discovers from scratch, and every
    probe rides sqlmap's own boolean inference (request / comparison / WAF stack)."""
    from lib.core.data import conf
    from lib.core.data import kb
    from lib.core.data import logger
    from lib.core.enums import CHARSET_TYPE
    from lib.core.enums import EXPECTED
    from lib.core.exception import SqlmapDataException
    from lib.core.exception import SqlmapUnsupportedFeatureException
    from lib.request.inject import checkBooleanExpression
    from lib.request.inject import getValue
    from plugins.generic.enumeration import Enumeration
    from plugins.generic.misc import Miscellaneous

    # boolean-blind-only oracle (no inband UNION marker): whole-page true/false, so a
    # reflective target that filters out the reflected marker can't defeat it
    def _blindOracle(condition):
        return getValue(condition, expected=EXPECTED.BOOL, charsetType=CHARSET_TYPE.BINARY,
                        suppressOutput=True, union=False, error=False, time=False)

    class _EsperantoHandler(Enumeration, Miscellaneous):
        def __init__(self):
            Enumeration.__init__(self)
            Miscellaneous.__init__(self)
            self._esp = None
            self._notesLogged = 0       # how many dialect.notes already surfaced (see _flushNotes)
            self._identCache = {}       # current user/db: fetch (and announce) once
            self._colCache = {}         # (db, table) -> ordered column names, so a dump
                                        # reuses what --columns already enumerated
            self._scopeCache = {}       # table -> resolved schema (see _scopeFor)

        def __getattr__(self, name):
            # this handler only mixes in Enumeration/Miscellaneous (read-only, boolean-oracle
            # extraction); any DBMS-specific capability (--os-*/--file-*/--sql-shell/--udf-*/--reg-*/
            # --cleanup) has no generic equivalent - fail cleanly instead of an AttributeError deep
            # in action(). Leading underscore excluded so internal/dunder lookups behave normally.
            if name.startswith("_"):
                raise AttributeError(name)
            raise SqlmapUnsupportedFeatureException("Esperanto does not support '%s' (DBMS-agnostic engine covers enumeration/dump only)" % name)

        def _engine(self):
            if self._esp is None:
                # esperanto is a PURE boolean-oracle engine: every probe is one true/false
                # question, so it gains nothing from UNION/error inband extraction - while
                # those need a concatenated marker whose generic form is CONCAT() when the
                # backend is unidentified (agent.py), and CONCAT() does not exist on SQLite/
                # Firebird/Oracle (they use ||) so every such probe errors. so PREFER the
                # boolean-blind technique: no marker, no concatenation, whole-page true/false,
                # works everywhere. only fall back to whatever-technique-is-available if the
                # target has no usable boolean-blind vector. _ask decides what an undecidable
                # probe MEANS by context (skip a candidate rung vs degrade a data read loudly).
                esp = Esperanto(_blindOracle, retries=2)
                logger.info("Esperanto is discovering the back-end SQL dialect (agnostic mode, boolean-blind)")
                try:
                    esp.discover()
                except RuntimeError:
                    # no usable boolean-blind vector on this target - retry with any technique
                    # sqlmap detected (UNION/error/time); may hit the CONCAT limitation above
                    esp = Esperanto(lambda condition: checkBooleanExpression(condition), retries=2)
                    logger.info("Esperanto retrying discovery via any available inference technique")
                    try:
                        esp.discover()
                    except RuntimeError as ex:
                        # genuinely unusable (unstable target, or no substring/pattern
                        # primitive) - stop cleanly instead of surfacing an internal traceback
                        raise SqlmapDataException("Esperanto could not establish a reliable extraction oracle on this target (%s)" % ex)
                logger.info("Esperanto dialect verdict: %s" % (esp.identify().get("product") or "unknown"))
                esp._progress = lambda value: logger.info("retrieved: %s" % _sanitize(value))  # live feedback (sanitized)
                self._esp = esp
                self._notesLogged = 0
                self._flushNotes()                      # discovery-time notes
            return self._esp

        def _flushNotes(self):
            # surface degradation notes LOUDLY as they accrue. Enumeration/dump append notes
            # AFTER discovery, so logging once would hide every runtime degradation (incomplete
            # listing, truncation, blocked paging) - flush the NEW ones after each operation.
            notes = self._esp.dialect.notes if self._esp else []
            for note in notes[self._notesLogged:]:
                logger.warning("Esperanto: %s" % note)
            self._notesLogged = len(notes)

        def _scopeDb(self):
            # the database to scope table/column lookups to: -D if given, else the
            # current one. WITHOUT this, a same-named table in another schema (e.g.
            # information_schema.USERS vs shop.users) merges columns and breaks dump.
            return conf.db or self.getCurrentDb()

        def _scopeFor(self, table):
            # scope for a SPECIFIC table: -D wins; else the current schema IF the table
            # is there; else the schema the table actually lives in (PG-family: tables
            # often sit in 'public' while current_schema is the login user's own schema).
            if conf.db:
                return conf.db
            if table in self._scopeCache:
                return self._scopeCache[table]
            esp = self._engine()
            cur = self.getCurrentDb()
            scope = cur if (cur and esp.hasTable(table, cur)) else (esp.tableSchema(table) or cur)
            self._scopeCache[table] = scope
            return scope

        def _db(self):
            return self._scopeDb() or "<current>"

        def getFingerprint(self):
            # concise fingerprint only; the version banner is shown for --banner, not
            # printed unbidden on every run (and not re-extracted here)
            product = self._engine().identify().get("product") or "unknown"
            return "back-end DBMS: %s (via Esperanto DBMS-agnostic engine)" % product

        def getBanner(self):
            # the ONLY path that blind-reads the full version string (expensive); the
            # fingerprint/product naming never does
            logger.info("fetching banner")
            kb.data.banner = self._engine().banner()
            return kb.data.banner

        def getCurrentUser(self):
            if "user" not in self._identCache:
                expr = self._engine().dialect.identity.get("user")
                if expr:
                    logger.info("fetching current user")
                self._identCache["user"] = self._safeExtract(expr) if expr else None
            kb.data.currentUser = self._identCache["user"]
            return kb.data.currentUser

        def getCurrentDb(self):
            # called repeatedly to scope tables/columns/dump -> fetch and announce once
            if "db" not in self._identCache:
                expr = self._engine().dialect.identity.get("database")
                if expr:
                    logger.info("fetching current database")
                self._identCache["db"] = self._safeExtract(expr) if expr else None
            kb.data.currentDb = self._identCache["db"]
            return kb.data.currentDb

        def _safeExtract(self, expr):
            # current user/db are used as SQL QUALIFIERS + cache keys + scoping decisions, so
            # they must be EXACT - a truncated/case-ambiguous value here becomes wrong SQL.
            try:
                res = self._esp.extractResult(expr)
            except OracleUndecided:
                logger.warning("Esperanto could not retrieve %s (oracle undecided)" % expr)
                return None
            if res.value is not None and not res.exact:
                logger.warning("Esperanto: %s not recovered exactly (%s) - not used for scoping" % (expr, res.integrity))
                return None
            return res.value

        def isDba(self, user=None):
            # UNKNOWN, not a negative claim: Esperanto has no generic DBA probe, so returning
            # False would assert "not a DBA" on no evidence. Report it can't tell and return
            # None (unknown) so a transient can't be read as a proven privilege verdict.
            logger.warning("Esperanto cannot determine DBA status (no generic privilege probe)")
            kb.data.isDba = None
            return kb.data.isDba

        def getDbs(self):
            logger.info("fetching database names")
            kb.data.cachedDbs = self._engine().enumerate("database", limit=(conf.limitStop or 50)) or []
            self._flushNotes()
            return kb.data.cachedDbs

        def getTables(self, bruteForce=None):
            # scope to the requested database (-D) or the current one, so the listing
            # isn't polluted with every schema's tables (e.g. information_schema)
            db = conf.db or self.getCurrentDb()
            lim = conf.limitStop or 100
            names = self._engine().enumerate("table", limit=lim, schema=db) or []
            if not names and not conf.db and db != "public":
                # current schema empty (PG-family: login-user schema) -> tables usually
                # live in 'public'; broaden rather than report nothing
                pub = self._engine().enumerate("table", limit=lim, schema="public") or []
                if pub:
                    names, db = pub, "public"
            infoMsg = "fetching tables"
            if db:
                infoMsg += " for database '%s'" % db
            logger.info(infoMsg)
            kb.data.cachedTables = {db or "<current>": names}
            self._flushNotes()
            return kb.data.cachedTables

        def getColumns(self, onlyColNames=False, colTuple=None, bruteForce=None, dumpMode=False):
            if not conf.tbl:
                logger.error("Esperanto needs a table (-T) to enumerate columns")
                return {}
            db = self._scopeFor(conf.tbl)
            infoMsg = "fetching columns for table '%s'" % conf.tbl
            if db:
                infoMsg += " in database '%s'" % db
            logger.info(infoMsg)
            names = self._engine().columns(conf.tbl, schema=db) or []
            self._colCache[(db, conf.tbl)] = names       # let a following dump reuse these
            kb.data.cachedColumns = {db or "<current>": {conf.tbl: dict((n, None) for n in names)}}
            self._flushNotes()
            return kb.data.cachedColumns

        def getSchema(self):
            esp = self._engine()
            # use the EFFECTIVE db that getTables() actually resolved (it may have broadened to
            # 'public' when the current schema was empty) - recomputing _db() here would miss
            # that and look columns up in the wrong (empty) schema.
            tabmap = self.getTables()
            effdb, tables = next(iter(tabmap.items()), ("<current>", []))
            colscope = self._scopeDb() if effdb == "<current>" else effdb
            schema = {}
            for table in (tables or []):
                schema[table] = dict((n, None) for n in (esp.columns(table, schema=colscope) or []))
            kb.data.cachedColumns = {effdb: schema}
            self._flushNotes()
            return kb.data.cachedColumns

        def dumpTable(self, foundData=None):
            if not conf.tbl:
                logger.error("Esperanto needs a table (-T) to dump")
                return
            db = self._scopeFor(conf.tbl)
            cols = [c.strip() for c in conf.col.split(",")] if conf.col else None
            if cols is None:
                cols = self._colCache.get((db, conf.tbl))    # reuse --columns' result; don't re-walk
            infoMsg = "fetching entries"
            if cols:
                infoMsg += " of column(s) '%s'" % ", ".join(cols)
            infoMsg += " for table '%s'" % conf.tbl
            if db:
                infoMsg += " in database '%s'" % db
            logger.info(infoMsg)
            # sqlmap row-SELECTORS change WHICH rows come back, so REFUSE (don't silently return
            # different data): --where filters, --start offsets. --stop is honored as a row cap.
            if getattr(conf, "dumpWhere", None):
                logger.error("Esperanto cannot honor --where; refusing rather than returning unfiltered rows")
                return
            if getattr(conf, "limitStart", None):
                logger.error("Esperanto cannot honor --start; refusing rather than returning the wrong row range")
                return
            # the SAME qualified reference for count and dump, so a non-default schema can't make
            # the count hit a different (unqualified) table than the dump (#8).
            qtable = self._engine().qualify(conf.tbl, db)
            if conf.limitStop:
                limit = conf.limitStop
            else:
                try:
                    limit = self._engine().extractInteger("(SELECT COUNT(*) FROM %s)" % qtable) or 10
                except (OracleUndecided, OverflowError):
                    limit = 1 << 30             # unknown count -> effectively "all" (keyset stops at end)
                logger.info("no --stop given; dumping all %s rows" % (limit if limit < (1 << 30) else "(count unknown)"))
            result = self._engine().dump(conf.tbl, columns=cols, schema=db, limit=limit)
            if not result or not result["columns"]:
                logger.error("Esperanto could not dump table '%s'" % conf.tbl)
                return
            table_data = {}
            for i, name in enumerate(result["columns"]):
                values = [("NULL" if row[i] is None else row[i]) for row in result["rows"]]
                width = max([len(name)] + [len(v) for v in values]) if values else len(name)
                table_data[name] = {"length": width, "values": values}
            table_data["__infos__"] = {"count": len(result["rows"]), "table": conf.tbl, "db": self._db()}
            if not result["complete"]:
                logger.warning("Esperanto dump of '%s' may be incomplete" % conf.tbl)
            if not result.get("exact", True):       # coverage may be complete yet content inexact
                logger.warning("Esperanto dump of '%s': values recovered via a collation-dependent "
                               "comparison - case/accents may not be byte-exact" % conf.tbl)
            kb.data.dumpedTable = table_data
            self._flushNotes()
            conf.dumper.dbTableValues(kb.data.dumpedTable)

    return _EsperantoHandler()
