Geforce File Manager

NAME	SIZE	ACTIONS
📁 __pycache__		[ DEL ]
📁 cli		[ DEL ]
📄 __init__.py	1,559 B	[ EDIT ] \| [ DEL ]
📄 big5freq.py	31,254 B	[ EDIT ] \| [ DEL ]
📄 big5prober.py	1,757 B	[ EDIT ] \| [ DEL ]
📄 chardistribution.py	9,411 B	[ EDIT ] \| [ DEL ]
📄 charsetgroupprober.py	3,787 B	[ EDIT ] \| [ DEL ]
📄 charsetprober.py	5,110 B	[ EDIT ] \| [ DEL ]
📄 codingstatemachine.py	3,590 B	[ EDIT ] \| [ DEL ]
📄 compat.py	1,134 B	[ EDIT ] \| [ DEL ]
📄 cp949prober.py	1,855 B	[ EDIT ] \| [ DEL ]
📄 enums.py	1,661 B	[ EDIT ] \| [ DEL ]
📄 escprober.py	3,950 B	[ EDIT ] \| [ DEL ]
📄 escsm.py	10,510 B	[ EDIT ] \| [ DEL ]
📄 eucjpprober.py	3,749 B	[ EDIT ] \| [ DEL ]
📄 euckrfreq.py	13,546 B	[ EDIT ] \| [ DEL ]
📄 euckrprober.py	1,748 B	[ EDIT ] \| [ DEL ]
📄 euctwfreq.py	31,621 B	[ EDIT ] \| [ DEL ]
📄 euctwprober.py	1,747 B	[ EDIT ] \| [ DEL ]
📄 gb2312freq.py	20,715 B	[ EDIT ] \| [ DEL ]
📄 gb2312prober.py	1,754 B	[ EDIT ] \| [ DEL ]
📄 hebrewprober.py	13,838 B	[ EDIT ] \| [ DEL ]
📄 jisfreq.py	25,777 B	[ EDIT ] \| [ DEL ]
📄 jpcntx.py	19,643 B	[ EDIT ] \| [ DEL ]
📄 langbulgarianmodel.py	12,839 B	[ EDIT ] \| [ DEL ]
📄 langcyrillicmodel.py	17,948 B	[ EDIT ] \| [ DEL ]
📄 langgreekmodel.py	12,688 B	[ EDIT ] \| [ DEL ]
📄 langhebrewmodel.py	11,345 B	[ EDIT ] \| [ DEL ]
📄 langhungarianmodel.py	12,592 B	[ EDIT ] \| [ DEL ]
📄 langthaimodel.py	11,290 B	[ EDIT ] \| [ DEL ]
📄 langturkishmodel.py	11,102 B	[ EDIT ] \| [ DEL ]
📄 latin1prober.py	5,370 B	[ EDIT ] \| [ DEL ]
📄 mbcharsetprober.py	3,413 B	[ EDIT ] \| [ DEL ]
📄 mbcsgroupprober.py	2,012 B	[ EDIT ] \| [ DEL ]
📄 mbcssm.py	25,481 B	[ EDIT ] \| [ DEL ]
📄 sbcharsetprober.py	5,657 B	[ EDIT ] \| [ DEL ]
📄 sbcsgroupprober.py	3,546 B	[ EDIT ] \| [ DEL ]
📄 sjisprober.py	3,774 B	[ EDIT ] \| [ DEL ]
📄 universaldetector.py	12,485 B	[ EDIT ] \| [ DEL ]
📄 utf8prober.py	2,766 B	[ EDIT ] \| [ DEL ]
📄 version.py	242 B	[ EDIT ] \| [ DEL ]

[ CLOSE ]

EDIT: mbcharsetprober.py

######################## BEGIN LICENSE BLOCK ########################
# The Original Code is Mozilla Universal charset detector code.
#
# The Initial Developer of the Original Code is
# Netscape Communications Corporation.
# Portions created by the Initial Developer are Copyright (C) 2001
# the Initial Developer. All Rights Reserved.
#
# Contributor(s):
#   Mark Pilgrim - port to Python
#   Shy Shalom - original C code
#   Proofpoint, Inc.
#
# This library is free software; you can redistribute it and/or
# modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version.
#
# This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# Lesser General Public License for more details.
#
# You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301  USA
######################### END LICENSE BLOCK #########################

from .charsetprober import CharSetProber
from .enums import ProbingState, MachineState

class MultiByteCharSetProber(CharSetProber):
    """
    MultiByteCharSetProber
    """

def __init__(self, lang_filter=None):
        super(MultiByteCharSetProber, self).__init__(lang_filter=lang_filter)
        self.distribution_analyzer = None
        self.coding_sm = None
        self._last_char = [0, 0]

def reset(self):
        super(MultiByteCharSetProber, self).reset()
        if self.coding_sm:
            self.coding_sm.reset()
        if self.distribution_analyzer:
            self.distribution_analyzer.reset()
        self._last_char = [0, 0]

@property
    def charset_name(self):
        raise NotImplementedError

@property
    def language(self):
        raise NotImplementedError

def feed(self, byte_str):
        for i in range(len(byte_str)):
            coding_state = self.coding_sm.next_state(byte_str[i])
            if coding_state == MachineState.ERROR:
                self.logger.debug('%s %s prober hit error at byte %s',
                                  self.charset_name, self.language, i)
                self._state = ProbingState.NOT_ME
                break
            elif coding_state == MachineState.ITS_ME:
                self._state = ProbingState.FOUND_IT
                break
            elif coding_state == MachineState.START:
                char_len = self.coding_sm.get_current_charlen()
                if i == 0:
                    self._last_char[1] = byte_str[0]
                    self.distribution_analyzer.feed(self._last_char, char_len)
                else:
                    self.distribution_analyzer.feed(byte_str[i - 1:i + 1],
                                                    char_len)

self._last_char[0] = byte_str[-1]

if self.state == ProbingState.DETECTING:
            if (self.distribution_analyzer.got_enough_data() and
                    (self.get_confidence() > self.SHORTCUT_THRESHOLD)):
                self._state = ProbingState.FOUND_IT

return self.state

def get_confidence(self):
        return self.distribution_analyzer.get_confidence()