cpython/Demo/scripts/markov.py

#! /usr/bin/env python

class Markov:
    def __init__(self, histsize, choice):
        self.histsize = histsize
        self.choice = choice
        self.trans = {}
    def add(self, state, next):
        if state not in self.trans:
            self.trans[state] = [next]
        else:
            self.trans[state].append(next)
    def put(self, seq):
        n = self.histsize
        add = self.add
        add(None, seq[:0])
        for i in range(len(seq)):
            add(seq[max(0, i-n):i], seq[i:i+1])
        add(seq[len(seq)-n:], None)
    def get(self):
        choice = self.choice
        trans = self.trans
        n = self.histsize
        seq = choice(trans[None])
        while 1:
            subseq = seq[max(0, len(seq)-n):]
            options = trans[subseq]
            next = choice(options)
            if not next: break
            seq = seq + next
        return seq

def test():
    import sys, string, random, getopt
    args = sys.argv[1:]
    try:
        opts, args = getopt.getopt(args, '0123456789cdw')
    except getopt.error:
        print('Usage: markov [-#] [-cddqw] [file] ...')
        print('Options:')
        print('-#: 1-digit history size (default 2)')
        print('-c: characters (default)')
        print('-w: words')
        print('-d: more debugging output')
        print('-q: no debugging output')
        print('Input files (default stdin) are split in paragraphs')
        print('separated blank lines and each paragraph is split')
        print('in words by whitespace, then reconcatenated with')
        print('exactly one space separating words.')
        print('Output consists of paragraphs separated by blank')
        print('lines, where lines are no longer than 72 characters.')
    histsize = 2
    do_words = 0
    debug = 1
    for o, a in opts:
        if '-0' <= o <= '-9': histsize = eval(o[1:])
        if o == '-c': do_words = 0
        if o == '-d': debug = debug + 1
        if o == '-q': debug = 0
        if o == '-w': do_words = 1
    if not args: args = ['-']
    m = Markov(histsize, random.choice)
    try:
        for filename in args:
            if filename == '-':
                f = sys.stdin
                if f.isatty():
                    print('Sorry, need stdin from file')
                    continue
            else:
                f = open(filename, 'r')
            if debug: print('processing', filename, '...')
            text = f.read()
            f.close()
            paralist = string.splitfields(text, '\n\n')
            for para in paralist:
                if debug > 1: print('feeding ...')
                words = string.split(para)
                if words:
                    if do_words: data = tuple(words)
                    else: data = string.joinfields(words, ' ')
                    m.put(data)
    except KeyboardInterrupt:
        print('Interrupted -- continue with data read so far')
    if not m.trans:
        print('No valid input files')
        return
    if debug: print('done.')
    if debug > 1:
        for key in m.trans.keys():
            if key is None or len(key) < histsize:
                print(repr(key), m.trans[key])
        if histsize == 0: print(repr(''), m.trans[''])
        print()
    while 1:
        data = m.get()
        if do_words: words = data
        else: words = string.split(data)
        n = 0
        limit = 72
        for w in words:
            if n + len(w) > limit:
                print()
                n = 0
            print(w, end=' ')
            n = n + len(w) + 1
        print()
        print()

def tuple(list):
    if len(list) == 0: return ()
    if len(list) == 1: return (list[0],)
    i = len(list)//2
    return tuple(list[:i]) + tuple(list[i:])

if __name__ == "__main__":
    test()
/usr/local/bin/python -> /usr/bin/env python 1996-11-27 15:52:01 -04:00			`#! /usr/bin/env python`
Initial revision 1993-12-14 06:08:02 -04:00
			`class Markov:`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`def __init__(self, histsize, choice):`
			`self.histsize = histsize`
			`self.choice = choice`
			`self.trans = {}`
			`def add(self, state, next):`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`if state not in self.trans:`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`self.trans[state] = [next]`
			`else:`
			`self.trans[state].append(next)`
			`def put(self, seq):`
			`n = self.histsize`
			`add = self.add`
			`add(None, seq[:0])`
			`for i in range(len(seq)):`
			`add(seq[max(0, i-n):i], seq[i:i+1])`
			`add(seq[len(seq)-n:], None)`
			`def get(self):`
			`choice = self.choice`
			`trans = self.trans`
			`n = self.histsize`
			`seq = choice(trans[None])`
			`while 1:`
			`subseq = seq[max(0, len(seq)-n):]`
			`options = trans[subseq]`
			`next = choice(options)`
			`if not next: break`
			`seq = seq + next`
			`return seq`
Initial revision 1993-12-14 06:08:02 -04:00
			`def test():`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`import sys, string, random, getopt`
			`args = sys.argv[1:]`
			`try:`
			`opts, args = getopt.getopt(args, '0123456789cdw')`
			`except getopt.error:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print('Usage: markov [-#] [-cddqw] [file] ...')`
			`print('Options:')`
			`print('-#: 1-digit history size (default 2)')`
			`print('-c: characters (default)')`
			`print('-w: words')`
			`print('-d: more debugging output')`
			`print('-q: no debugging output')`
			`print('Input files (default stdin) are split in paragraphs')`
			`print('separated blank lines and each paragraph is split')`
			`print('in words by whitespace, then reconcatenated with')`
			`print('exactly one space separating words.')`
			`print('Output consists of paragraphs separated by blank')`
			`print('lines, where lines are no longer than 72 characters.')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`histsize = 2`
			`do_words = 0`
			`debug = 1`
			`for o, a in opts:`
			`if '-0' <= o <= '-9': histsize = eval(o[1:])`
			`if o == '-c': do_words = 0`
			`if o == '-d': debug = debug + 1`
			`if o == '-q': debug = 0`
			`if o == '-w': do_words = 1`
			`if not args: args = ['-']`
			`m = Markov(histsize, random.choice)`
			`try:`
			`for filename in args:`
			`if filename == '-':`
			`f = sys.stdin`
			`if f.isatty():`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print('Sorry, need stdin from file')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`continue`
			`else:`
			`f = open(filename, 'r')`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`if debug: print('processing', filename, '...')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`text = f.read()`
			`f.close()`
			`paralist = string.splitfields(text, '\n\n')`
			`for para in paralist:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`if debug > 1: print('feeding ...')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`words = string.split(para)`
			`if words:`
			`if do_words: data = tuple(words)`
			`else: data = string.joinfields(words, ' ')`
			`m.put(data)`
			`except KeyboardInterrupt:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print('Interrupted -- continue with data read so far')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`if not m.trans:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print('No valid input files')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`return`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`if debug: print('done.')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`if debug > 1:`
remove most uses of list(somedict.keys()) in Demo scripts 2007-08-06 18:07:53 -03:00			`for key in m.trans.keys():`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`if key is None or len(key) < histsize:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print(repr(key), m.trans[key])`
			`if histsize == 0: print(repr(''), m.trans[''])`
			`print()`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`while 1:`
			`data = m.get()`
			`if do_words: words = data`
			`else: words = string.split(data)`
			`n = 0`
			`limit = 72`
			`for w in words:`
			`if n + len(w) > limit:`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print()`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`n = 0`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print(w, end=' ')`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`n = n + len(w) + 1`
Run 2to3 over the Demo/ directory to shut up parse errors from 2to3 about lingering print statements. 2007-07-17 17:59:35 -03:00			`print()`
			`print()`
Initial revision 1993-12-14 06:08:02 -04:00
			`def tuple(list):`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`if len(list) == 0: return ()`
			`if len(list) == 1: return (list[0],)`
Merged revisions 66394,66404,66412,66414,66424-66436 via svnmerge from svn+ssh://pythondev@svn.python.org/python/trunk ........ r66394 \| benjamin.peterson \| 2008-09-11 17:04:02 -0500 (Thu, 11 Sep 2008) \| 1 line fix typo ........ r66404 \| gerhard.haering \| 2008-09-12 08:54:06 -0500 (Fri, 12 Sep 2008) \| 2 lines sqlite3 module: Mark iterdump() method as "Non-standard" like all the other methods not found in DB-API. ........ r66412 \| gerhard.haering \| 2008-09-12 13:58:57 -0500 (Fri, 12 Sep 2008) \| 2 lines Fixes issue #3103. In the sqlite3 module, made one more function static. All renaming public symbos now have the pysqlite prefix to avoid name clashes. This at least once created problems where the same symbol name appeared somewhere in Apache and the sqlite3 module was used from mod_python. ........ r66414 \| gerhard.haering \| 2008-09-12 17:33:22 -0500 (Fri, 12 Sep 2008) \| 2 lines Issue #3846: Release GIL during calls to sqlite3_prepare. This improves concurrent access to the same database file from multiple threads/processes. ........ r66424 \| andrew.kuchling \| 2008-09-12 20:22:08 -0500 (Fri, 12 Sep 2008) \| 1 line #687648 from Robert Schuppenies: use classic division. (RM Barry gave permission to update the demos.) ........ r66425 \| andrew.kuchling \| 2008-09-12 20:27:33 -0500 (Fri, 12 Sep 2008) \| 1 line #687648 from Robert Schuppenies: use classic division. From me: don't use string exception; flush stdout after printing ........ r66426 \| andrew.kuchling \| 2008-09-12 20:34:41 -0500 (Fri, 12 Sep 2008) \| 1 line #687648 from Robert Schuppenies: use classic division. From me: don't use string exception; add __main__ section ........ r66427 \| andrew.kuchling \| 2008-09-12 20:42:55 -0500 (Fri, 12 Sep 2008) \| 1 line #687648 from Robert Schuppenies: use classic division. From me: remove two stray semicolons ........ r66428 \| andrew.kuchling \| 2008-09-12 20:43:28 -0500 (Fri, 12 Sep 2008) \| 1 line #687648 from Robert Schuppenies: use classic division. ........ r66429 \| andrew.kuchling \| 2008-09-12 20:47:02 -0500 (Fri, 12 Sep 2008) \| 1 line Remove semicolon ........ r66430 \| andrew.kuchling \| 2008-09-12 20:48:36 -0500 (Fri, 12 Sep 2008) \| 1 line Subclass exception ........ r66431 \| andrew.kuchling \| 2008-09-12 20:56:56 -0500 (Fri, 12 Sep 2008) \| 1 line Fix SyntaxError ........ r66432 \| andrew.kuchling \| 2008-09-12 20:57:25 -0500 (Fri, 12 Sep 2008) \| 1 line Update uses of string exceptions ........ r66433 \| andrew.kuchling \| 2008-09-12 21:08:30 -0500 (Fri, 12 Sep 2008) \| 1 line Use title case ........ r66434 \| andrew.kuchling \| 2008-09-12 21:09:15 -0500 (Fri, 12 Sep 2008) \| 1 line Remove extra 'the'; the following title includes it ........ r66435 \| andrew.kuchling \| 2008-09-12 21:11:51 -0500 (Fri, 12 Sep 2008) \| 1 line #3288: Document as_integer_ratio ........ r66436 \| andrew.kuchling \| 2008-09-12 21:14:15 -0500 (Fri, 12 Sep 2008) \| 1 line Use title case ........ 2008-09-13 12:58:53 -03:00			`i = len(list)//2`
Whitespace normalization. Ran reindent.py over the entire source tree. 2004-07-18 02:56:09 -03:00			`return tuple(list[:i]) + tuple(list[i:])`
Initial revision 1993-12-14 06:08:02 -04:00
Add 'if __name__ == "__main__":' to files already as a usable as a module. 2004-09-11 13:34:35 -03:00			`if __name__ == "__main__":`
			`test()`