azure - ストアドプロシージャを使用した Azure documentdb 一括挿入

Question

こんにちは、16 個のコレクションを使用して、オブジェクトあたり 5 ～ 10k の範囲の約 3 ～ 400 万の json オブジェクトを挿入しています。ストアドプロシージャを使用してこれらのドキュメントを挿入しています。22 の容量ユニットがあります。

function bulkImport(docs) {
    var collection = getContext().getCollection();
    var collectionLink = collection.getSelfLink();

    // The count of imported docs, also used as current doc index.
    var count = 0;

    // Validate input.
    if (!docs) throw new Error("The array is undefined or null.");

    var docsLength = docs.length;
    if (docsLength == 0) {
        getContext().getResponse().setBody(0);
    }

    // Call the CRUD API to create a document.
    tryCreateOrUpdate(docs[count], callback);

    // Note that there are 2 exit conditions:
    // 1) The createDocument request was not accepted. 
    //    In this case the callback will not be called, we just call setBody and we are done.
    // 2) The callback was called docs.length times.
    //    In this case all documents were created and we don't need to call tryCreate anymore. Just call setBody and we are done.
    function tryCreateOrUpdate(doc, callback) {
        var isAccepted = true;
        var isFound = collection.queryDocuments(collectionLink, 'SELECT * FROM root r WHERE r.id = "' + doc.id + '"', function (err, feed, options) {
            if (err) throw err;
            if (!feed || !feed.length) {
                isAccepted = collection.createDocument(collectionLink, doc, callback);
            }
            else {
                // The metadata document.
                var existingDoc = feed[0];
                isAccepted = collection.replaceDocument(existingDoc._self, doc, callback);
            }
        });

        // If the request was accepted, callback will be called.
        // Otherwise report current count back to the client, 
        // which will call the script again with remaining set of docs.
        // This condition will happen when this stored procedure has been running too long
        // and is about to get cancelled by the server. This will allow the calling client
        // to resume this batch from the point we got to before isAccepted was set to false
        if (!isFound && !isAccepted) getContext().getResponse().setBody(count);
    }

    // This is called when collection.createDocument is done and the document has been persisted.
    function callback(err, doc, options) {
        if (err) throw err;

        // One more document has been inserted, increment the count.
        count++;

        if (count >= docsLength) {
            // If we have created all documents, we are done. Just set the response.
            getContext().getResponse().setBody(count);
        } else {
            // Create next document.
            tryCreateOrUpdate(docs[count], callback);
        }
    }

私のC＃コードは次のようになります

    public async Task<int> Add(List<JobDTO> entities)
            {

                    int currentCount = 0;
                    int documentCount = entities.Count;

                    while(currentCount < documentCount)
                    {
                        string argsJson = JsonConvert.SerializeObject(entities.Skip(currentCount).ToArray());
                        var args = new dynamic[] { JsonConvert.DeserializeObject<dynamic[]>(argsJson) };

                        // 6. execute the batch.
                        StoredProcedureResponse<int> scriptResult = await DocumentDBRepository.Client.ExecuteStoredProcedureAsync<int>(sproc.SelfLink, args);

                        // 7. Prepare for next batch.
                        int currentlyInserted = scriptResult.Response;

                        currentCount += currentlyInserted;

                    }

                    return currentCount;
            }

私が直面している問題は、挿入しようとする 400k のドキュメントのうち、エラーが発生せずにドキュメントが見逃されることがあります。

アプリケーションは、クラウド上にデプロイされた worker ロールです。documentDB に挿入するスレッドまたはインスタンスの数を増やすと、失われるドキュメントの数がはるかに多くなります。

問題が何であるかを把握する方法。事前に感謝します。

score 10 · Accepted Answer

このコードを試すと、docs.length で、長さが未定義であるというエラーが表示されることがわかりました。

function bulkImport(docs) {
    var collection = getContext().getCollection();
    var collectionLink = collection.getSelfLink();

    // The count of imported docs, also used as current doc index.
    var count = 0;

    // Validate input.
    if (!docs) throw new Error("The array is undefined or null.");

    var docsLength = docs.length; // length is undefined
}

多くのテスト (Azure ドキュメントには何も見つかりませんでした) の後、提案されたように配列を渡すことができないことに気付きました。パラメータはオブジェクトでなければなりませんでした。バッチコードを実行するには、このようにバッチコードを変更する必要がありました。

また、DocumentDB スクリプトエクスプローラー (入力ボックス) でドキュメントの配列を単純に渡すこともできないこともわかりました。プレースホルダーのヘルプテキストには、できると書かれていますが。

このコードは私のために働いた：

// psuedo object for reference only
docObject = {
  "items": [{doc}, {doc}, {doc}]
}

function bulkImport(docObject) {
    var context = getContext();
    var collection = context.getCollection();
    var collectionLink = collection.getSelfLink();
    var count = 0;

    // Check input
    if (!docObject.items || !docObject.items.length) throw new Error("invalid document input parameter or undefined.");
    var docs = docObject.items;
    var docsLength = docs.length;
    if (docsLength == 0) {
        context.getResponse().setBody(0);
    }

    // Call the funct to create a document.
    tryCreateOrUpdate(docs[count], callback);

    // Obviously I have truncated this function. The above code should help you understand what has to change.
}

Azure のドキュメントが追いつくか、見逃した場合に見つけやすくなることを願っています。

また、Azurites が更新されることを期待して、スクリプトエクスプローラーのバグレポートを投稿する予定です。

azure - ストアド プロシージャを使用した Azure documentdb 一括挿入

3 に答える 3

Related

Reference

azure - ストアドプロシージャを使用した Azure documentdb 一括挿入