Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); ERROR while pulling python execute.py by pumpkinband · Pull Request #95 · llSourcell/tensorflow_chatbot · GitHub
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,5 @@
data/
working_dir/
*.pyc
.venv/
.idea/
73 changes: 66 additions & 7 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,22 +20,81 @@ Use [pip](https://pypi.python.org/pypi/pip) to install any missing dependencies
Usage
===========

To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so
Create venv & install dependencies:

> Using viritualenv here to be compatable with python2, install with "pip install virtualenv"

```
# create venv
python -m virtualenv .venv

# enter venv (assuming macos/linux)
source .venv/bin/activate

# install requirements
pip install -r requirements.txt
```

Training
--------------------

0. Create directories
```
mkdir working_dir
mkdir data
```

1. Download the [Cornell Movie Dialogue dataset](https://www.cs.cornell.edu/~cristian/Cornell_Movie-Dialogs_Corpus.html) and place the unzipped content in the `data/` directory.
```
cd data
wget http://www.mpi-sws.org/~cristian/data/cornell_movie_dialogs_corpus.zip
unzip cornell_movie_dialogs_corpus.zip

# move to `data/` root
mv "corenell movie-dialogs corpus"/* .
```

2. Prepare data for training, in the `data/` directory, run the `prepare_data.py` script
```
# move to project root
cd ..
python prepare_data.py
```

3. To train the bot, edit the `seq2seq.ini` file so that mode is set to train like so

`mode = train`

then run the code like so
4. Start training, by running the code like so:

``python execute.py``

> There is no mechanism to stop training, you will need to 'ctrl-c' to stop training after a period of time.


Test
-------------

1. To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so

`mode = test`

``python execute.py``
2. To test run the code like so:

To test the bot during or after training, edit the `seq2seq.ini` file so that mode is set to test like so
```
python execute.py

`mode = test`
>> Mode : test

then run the code like so
Reading model parameters from working_dir/seq2seq.ckpt-10200
>
```

``python execute.py``

3. Confrim...
- What does "Test do?"
- How to use it?

Challenge
===========
Expand Down
154 changes: 79 additions & 75 deletions data_utils.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -20,16 +20,16 @@

import os
import re

from six.moves import urllib
from io import open
from collections import Counter

from tensorflow.python.platform import gfile

# Special vocabulary symbols - we always put them at the start.
_PAD = b"_PAD"
_GO = b"_GO"
_EOS = b"_EOS"
_UNK = b"_UNK"
_PAD = "_PAD"
_GO = "_GO"
_EOS = "_EOS"
_UNK = "_UNK"
_START_VOCAB = [_PAD, _GO, _EOS, _UNK]

PAD_ID = 0
Expand All@@ -38,93 +38,97 @@
UNK_ID = 3

# Regular expressions used to tokenize.
_WORD_SPLIT = re.compile(b"([.,!?\"':;)(])")
_DIGIT_RE = re.compile(br"\d")
_WORD_SPLIT = re.compile("([.,!?\"':;)(])")
_DIGIT_RE = re.compile(r"\d")

CORNELL_MOVIE_CORPUS_ENCODING = 'ISO-8859-2'


def basic_tokenizer(sentence):
"""Very basic tokenizer: split the sentence into a list of tokens."""
words = []
for space_separated_fragment in sentence.strip().split():
words.extend(re.split(_WORD_SPLIT, space_separated_fragment))
return [w for w in words if w]
"""Very basic tokenizer: split the sentence into a list of tokens."""
all_words = []
for space_separated_fragment in sentence.strip().split():
words = re.split(_WORD_SPLIT, space_separated_fragment)
for word in words:
if word:
all_words.append(word)
return all_words


def create_vocabulary(vocabulary_path, data_path, max_vocabulary_size,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
if not tokenizer:
tokenizer = basic_tokenizer

if not os.path.exists(vocabulary_path):
print("Creating vocabulary %s from %s" % (vocabulary_path, data_path))
vocab = Counter()
with open(data_path, 'rt', encoding='utf8') as f:
for counter, sentence in enumerate(f, 1):
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(sentence)
for w in tokens:
if normalize_digits:
word = re.sub(_DIGIT_RE, '0', w)
else:
word = w
vocab[word] += 1

vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :', len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
print('>>>> Vocab Truncated to: {}'.format(max_vocabulary_size))
with open(vocabulary_path, 'wt', encoding='utf8') as vocab_file:
for w in vocab_list:
vocab_file.write(w + '\n')


def initialize_vocabulary(vocabulary_path, encoding=CORNELL_MOVIE_CORPUS_ENCODING):
vocab = {}
with gfile.GFile(data_path, mode="rb") as f:
counter = 0
for line in f:
counter += 1
if counter % 100000 == 0:
print(" processing line %d" % counter)
tokens = tokenizer(line) if tokenizer else basic_tokenizer(line)
for w in tokens:
word = re.sub(_DIGIT_RE, b"0", w) if normalize_digits else w
if word in vocab:
vocab[word] += 1
else:
vocab[word] = 1
vocab_list = _START_VOCAB + sorted(vocab, key=vocab.get, reverse=True)
print('>> Full Vocabulary Size :',len(vocab_list))
if len(vocab_list) > max_vocabulary_size:
vocab_list = vocab_list[:max_vocabulary_size]
with gfile.GFile(vocabulary_path, mode="wb") as vocab_file:
for w in vocab_list:
vocab_file.write(w + b"\n")


def initialize_vocabulary(vocabulary_path):

if gfile.Exists(vocabulary_path):
rev_vocab = []
with gfile.GFile(vocabulary_path, mode="rb") as f:
rev_vocab.extend(f.readlines())
rev_vocab = [line.strip() for line in rev_vocab]
vocab = dict([(x, y) for (y, x) in enumerate(rev_vocab)])
if gfile.Exists(vocabulary_path):
with open(vocabulary_path, 'rt', encoding=encoding) as f:
for index, line in enumerate(f, 1):
element = line.strip()
rev_vocab.append(element)
vocab[element] = index
assert len(vocab) == len(rev_vocab)
if not (vocab and rev_vocab):
raise ValueError('File empty: {}'.format(vocabulary_path))
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)
return vocab, rev_vocab
else:
raise ValueError("Vocabulary file %s not found.", vocabulary_path)


def sentence_to_token_ids(sentence, vocabulary, tokenizer=None, normalize_digits=True):

if tokenizer:
if not tokenizer:
tokenizer = basic_tokenizer
words = tokenizer(sentence)
else:
words = basic_tokenizer(sentence)
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, b"0", w), UNK_ID) for w in words]
if not normalize_digits:
return [vocabulary.get(w, UNK_ID) for w in words]
# Normalize digits by 0 before looking words up in the vocabulary.
return [vocabulary.get(re.sub(_DIGIT_RE, '0', w), UNK_ID) for w in words]


def data_to_token_ids(data_path, target_path, vocabulary_path,
tokenizer=None, normalize_digits=True):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
counter = 0
for line in data_file:
counter += 1
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")



def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size, dec_vocabulary_size, tokenizer=None):

if not gfile.Exists(target_path):
print("Tokenizing data in %s" % data_path)
vocab, _ = initialize_vocabulary(vocabulary_path)
with gfile.GFile(data_path, mode="rb") as data_file:
with gfile.GFile(target_path, mode="w") as tokens_file:
for counter, line in enumerate(data_file, 1):
if counter % 100000 == 0:
print(" tokenizing line %d" % counter)
token_ids = sentence_to_token_ids(line, vocab, tokenizer,
normalize_digits)
tokens_file.write(" ".join([str(tok) for tok in token_ids]) + "\n")


def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_dec, enc_vocabulary_size,
dec_vocabulary_size, tokenizer=None):
# Create vocabularies of the appropriate sizes.
enc_vocab_path = os.path.join(working_directory, "vocab%d.enc" % enc_vocabulary_size)
dec_vocab_path = os.path.join(working_directory, "vocab%d.dec" % dec_vocabulary_size)
Expand All@@ -143,4 +147,4 @@ def prepare_custom_data(working_directory, train_enc, train_dec, test_enc, test_
data_to_token_ids(test_enc, enc_dev_ids_path, enc_vocab_path, tokenizer)
data_to_token_ids(test_dec, dec_dev_ids_path, dec_vocab_path, tokenizer)

return (enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path)
return enc_train_ids_path, dec_train_ids_path, enc_dev_ids_path, dec_dev_ids_path, enc_vocab_path, dec_vocab_path
Loading